Micron Document
<!DOCTYPE html>
<html class="client-nojs vector-feature-night-mode-disabled vector-feature-language-in-header-enabled vector-feature-language-in-main-page-header-disabled vector-feature-page-tools-pinned-disabled vector-feature-toc-pinned-clientpref-1 vector-feature-main-menu-pinned-disabled vector-feature-limited-width-clientpref-1 vector-feature-limited-width-content-enabled vector-feature-custom-font-size-clientpref-1 vector-feature-appearance-pinned-clientpref-1 vector-sticky-header-enabled" lang="en" dir="ltr"><head>
<meta charset="UTF-8">
<title>Multinomial logistic regression</title>
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<link rel="canonical" href="https://en.wikipedia.org/wiki/Multinomial_logistic_regression"> <link href="./mw/ext.cite.styles.css" rel="stylesheet" type="text/css">
<link href="./mw/ext.math.styles.css" rel="stylesheet" type="text/css">
<link href="./mw/skins.vector.icons.css" rel="stylesheet" type="text/css">
<link href="./mw/skins.vector.search.codex.styles.css" rel="stylesheet" type="text/css">
<link href="./mw/skins.vector.styles.css" rel="stylesheet" type="text/css">
<link href="./mw/user.styles.css" rel="stylesheet" type="text/css">
<meta name="ResourceLoaderDynamicStyles" content="">
<link rel="stylesheet" type="text/css" href="./mw/site.styles.css">
<link rel="stylesheet" type="text/css" href="./mw/noscript.css">
<link rel="stylesheet" type="text/css" href="./footer.css">
<link rel="stylesheet" type="text/css" href="./vector-2022.css">
</head>
<body class="skin--responsive skin-vector skin-vector-search-vue mediawiki ltr sitedir-ltr mw-hide-empty-elt ns-0 ns-subject page-Multinomial_logistic_regression rootpage-Multinomial_logistic_regression skin-vector-2022 action-view">
<div class="mw-page-container">
<div class="mw-page-container-inner">
<div class="mw-content-container">
<main id="content" class="mw-body">
<header class="mw-body-header vector-page-titlebar">
<h1 id="firstHeading" class="firstHeading mw-first-heading">
<span id="openzim-page-title" class="mw-page-title-main"><span class="mw-page-title-main">Multinomial logistic regression</span></span>
</h1>
</header>
<a id="top"></a>
<div id="bodyContent" class="vector-body ve-init-mw-desktopArticleTarget-targetContainer" aria-labelledby="firstHeading" data-mw-ve-target-container="">
<div id="mw-content-text" class="mw-body-content mw-content-ltr" lang="en" dir="ltr"><div class="mw-content-ltr mw-parser-output" lang="en" dir="ltr">
<style data-mw-deduplicate="TemplateStyles:r1236090951">
/* start https://en.wikipedia.org/ */


.mw-parser-output .hatnote{font-style:italic}.mw-parser-output div.hatnote{padding-left:1.6em;margin-bottom:0.5em}.mw-parser-output .hatnote i{font-style:normal}.mw-parser-output .hatnote+link+.hatnote{margin-top:-0.5em}@media print{body.ns-0 .mw-parser-output .hatnote{display:none!important}}


/* end https://en.wikipedia.org/ */
</style><div role="note" class="hatnote navigation-not-searchable">"Multinomial regression" redirects here. For the related Probit procedure, see <a href="Multinomial_probit" title="Multinomial probit">Multinomial probit</a>.</div>
<style data-mw-deduplicate="TemplateStyles:r1251242444">
/* start https://en.wikipedia.org/ */


.mw-parser-output .ambox{border:1px solid #a2a9b1;border-left:10px solid #36c;background-color:#fbfbfb;box-sizing:border-box}.mw-parser-output .ambox+link+.ambox,.mw-parser-output .ambox+link+style+.ambox,.mw-parser-output .ambox+link+link+.ambox,.mw-parser-output .ambox+.mw-empty-elt+link+.ambox,.mw-parser-output .ambox+.mw-empty-elt+link+style+.ambox,.mw-parser-output .ambox+.mw-empty-elt+link+link+.ambox{margin-top:-1px}html body.mediawiki .mw-parser-output .ambox.mbox-small-left{margin:4px 1em 4px 0;overflow:hidden;width:238px;border-collapse:collapse;font-size:88%;line-height:1.25em}.mw-parser-output .ambox-speedy{border-left:10px solid #b32424;background-color:#fee7e6}.mw-parser-output .ambox-delete{border-left:10px solid #b32424}.mw-parser-output .ambox-content{border-left:10px solid #f28500}.mw-parser-output .ambox-style{border-left:10px solid #fc3}.mw-parser-output .ambox-move{border-left:10px solid #9932cc}.mw-parser-output .ambox-protection{border-left:10px solid #a2a9b1}.mw-parser-output .ambox .mbox-text{border:none;padding:0.25em 0.5em;width:100%}.mw-parser-output .ambox .mbox-image{border:none;padding:2px 0 2px 0.5em;text-align:center}.mw-parser-output .ambox .mbox-imageright{border:none;padding:2px 0.5em 2px 0;text-align:center}.mw-parser-output .ambox .mbox-empty-cell{border:none;padding:0;width:1px}.mw-parser-output .ambox .mbox-image-div{width:52px}@media(min-width:720px){.mw-parser-output .ambox{margin:0 10%}}@media print{body.ns-0 .mw-parser-output .ambox{display:none!important}}


/* end https://en.wikipedia.org/ */
</style>
<style data-mw-deduplicate="TemplateStyles:r1129693374">
/* start https://en.wikipedia.org/ */


.mw-parser-output .hlist dl,.mw-parser-output .hlist ol,.mw-parser-output .hlist ul{margin:0;padding:0}.mw-parser-output .hlist dd,.mw-parser-output .hlist dt,.mw-parser-output .hlist li{margin:0;display:inline}.mw-parser-output .hlist.inline,.mw-parser-output .hlist.inline dl,.mw-parser-output .hlist.inline ol,.mw-parser-output .hlist.inline ul,.mw-parser-output .hlist dl dl,.mw-parser-output .hlist dl ol,.mw-parser-output .hlist dl ul,.mw-parser-output .hlist ol dl,.mw-parser-output .hlist ol ol,.mw-parser-output .hlist ol ul,.mw-parser-output .hlist ul dl,.mw-parser-output .hlist ul ol,.mw-parser-output .hlist ul ul{display:inline}.mw-parser-output .hlist .mw-empty-li{display:none}.mw-parser-output .hlist dt::after{content:": "}.mw-parser-output .hlist dd::after,.mw-parser-output .hlist li::after{content:" · ";font-weight:bold}.mw-parser-output .hlist dd:last-child::after,.mw-parser-output .hlist dt:last-child::after,.mw-parser-output .hlist li:last-child::after{content:none}.mw-parser-output .hlist dd dd:first-child::before,.mw-parser-output .hlist dd dt:first-child::before,.mw-parser-output .hlist dd li:first-child::before,.mw-parser-output .hlist dt dd:first-child::before,.mw-parser-output .hlist dt dt:first-child::before,.mw-parser-output .hlist dt li:first-child::before,.mw-parser-output .hlist li dd:first-child::before,.mw-parser-output .hlist li dt:first-child::before,.mw-parser-output .hlist li li:first-child::before{content:" (";font-weight:normal}.mw-parser-output .hlist dd dd:last-child::after,.mw-parser-output .hlist dd dt:last-child::after,.mw-parser-output .hlist dd li:last-child::after,.mw-parser-output .hlist dt dd:last-child::after,.mw-parser-output .hlist dt dt:last-child::after,.mw-parser-output .hlist dt li:last-child::after,.mw-parser-output .hlist li dd:last-child::after,.mw-parser-output .hlist li dt:last-child::after,.mw-parser-output .hlist li li:last-child::after{content:")";font-weight:normal}.mw-parser-output .hlist ol{counter-reset:listitem}.mw-parser-output .hlist ol>li{counter-increment:listitem}.mw-parser-output .hlist ol>li::before{content:" "counter(listitem)"\a0 "}.mw-parser-output .hlist dd ol>li:first-child::before,.mw-parser-output .hlist dt ol>li:first-child::before,.mw-parser-output .hlist li ol>li:first-child::before{content:" ("counter(listitem)"\a0 "}


/* end https://en.wikipedia.org/ */
</style><style data-mw-deduplicate="TemplateStyles:r1246091330">
/* start https://en.wikipedia.org/ */


.mw-parser-output .sidebar{width:22em;float:right;clear:right;margin:0.5em 0 1em 1em;background:var(--background-color-neutral-subtle,#f8f9fa);border:1px solid var(--border-color-base,#a2a9b1);padding:0.2em;text-align:center;line-height:1.4em;font-size:88%;border-collapse:collapse;display:table}body.skin-minerva .mw-parser-output .sidebar{display:table!important;float:right!important;margin:0.5em 0 1em 1em!important}.mw-parser-output .sidebar-subgroup{width:100%;margin:0;border-spacing:0}.mw-parser-output .sidebar-left{float:left;clear:left;margin:0.5em 1em 1em 0}.mw-parser-output .sidebar-none{float:none;clear:both;margin:0.5em 1em 1em 0}.mw-parser-output .sidebar-outer-title{padding:0 0.4em 0.2em;font-size:125%;line-height:1.2em;font-weight:bold}.mw-parser-output .sidebar-top-image{padding:0.4em}.mw-parser-output .sidebar-top-caption,.mw-parser-output .sidebar-pretitle-with-top-image,.mw-parser-output .sidebar-caption{padding:0.2em 0.4em 0;line-height:1.2em}.mw-parser-output .sidebar-pretitle{padding:0.4em 0.4em 0;line-height:1.2em}.mw-parser-output .sidebar-title,.mw-parser-output .sidebar-title-with-pretitle{padding:0.2em 0.8em;font-size:145%;line-height:1.2em}.mw-parser-output .sidebar-title-with-pretitle{padding:0.1em 0.4em}.mw-parser-output .sidebar-image{padding:0.2em 0.4em 0.4em}.mw-parser-output .sidebar-heading{padding:0.1em 0.4em}.mw-parser-output .sidebar-content{padding:0 0.5em 0.4em}.mw-parser-output .sidebar-content-with-subgroup{padding:0.1em 0.4em 0.2em}.mw-parser-output .sidebar-above,.mw-parser-output .sidebar-below{padding:0.3em 0.8em;font-weight:bold}.mw-parser-output .sidebar-collapse .sidebar-above,.mw-parser-output .sidebar-collapse .sidebar-below{border-top:1px solid #aaa;border-bottom:1px solid #aaa}.mw-parser-output .sidebar-navbar{text-align:right;font-size:115%;padding:0 0.4em 0.4em}.mw-parser-output .sidebar-list-title{padding:0 0.4em;text-align:left;font-weight:bold;line-height:1.6em;font-size:105%}.mw-parser-output .sidebar-list-title-c{padding:0 0.4em;text-align:center;margin:0 3.3em}@media(max-width:640px){body.mediawiki .mw-parser-output .sidebar{width:100%!important;clear:both;float:none!important;margin-left:0!important;margin-right:0!important}}body.skin--responsive .mw-parser-output .sidebar a>img{max-width:none!important}@media screen{html.skin-theme-clientpref-night .mw-parser-output .sidebar:not(.notheme) .sidebar-list-title,html.skin-theme-clientpref-night .mw-parser-output .sidebar:not(.notheme) .sidebar-title-with-pretitle{background:transparent!important}html.skin-theme-clientpref-night .mw-parser-output .sidebar:not(.notheme) .sidebar-title-with-pretitle a{color:var(--color-progressive)!important}}@media screen and (prefers-color-scheme:dark){html.skin-theme-clientpref-os .mw-parser-output .sidebar:not(.notheme) .sidebar-list-title,html.skin-theme-clientpref-os .mw-parser-output .sidebar:not(.notheme) .sidebar-title-with-pretitle{background:transparent!important}html.skin-theme-clientpref-os .mw-parser-output .sidebar:not(.notheme) .sidebar-title-with-pretitle a{color:var(--color-progressive)!important}}@media print{body.ns-0 .mw-parser-output .sidebar{display:none!important}}


/* end https://en.wikipedia.org/ */
</style><table class="sidebar nomobile nowraplinks hlist"><tbody><tr><td class="sidebar-pretitle">Part of a series on</td></tr><tr><th class="sidebar-title-with-pretitle"><a href="Regression_analysis" title="Regression analysis">Regression analysis</a></th></tr><tr><th class="sidebar-heading">
Models</th></tr><tr><td class="sidebar-content">
<ul><li><a href="Linear_regression" title="Linear regression">Linear regression</a></li>
<li><a href="Simple_linear_regression" title="Simple linear regression">Simple regression</a></li>
<li><a href="Polynomial_regression" title="Polynomial regression">Polynomial regression</a></li>
<li><a href="General_linear_model" title="General linear model">General linear model</a></li></ul></td>
</tr><tr><td class="sidebar-content">
<ul><li><a href="Generalized_linear_model" title="Generalized linear model">Generalized linear model</a></li>
<li><a href="Vector_generalized_linear_model" title="Vector generalized linear model">Vector generalized linear model</a></li>
<li><a href="Discrete_choice" title="Discrete choice">Discrete choice</a></li>
<li><a href="Binomial_regression" title="Binomial regression">Binomial regression</a></li>
<li><a href="Binary_regression" title="Binary regression">Binary regression</a></li>
<li><a href="Logistic_regression" title="Logistic regression">Logistic regression</a></li>

<li><a href="Mixed_logit" title="Mixed logit">Mixed logit</a></li>
<li><a href="Probit_model" title="Probit model">Probit</a></li>
<li><a href="Multinomial_probit" title="Multinomial probit">Multinomial probit</a></li>
<li><a href="Ordered_logit" title="Ordered logit">Ordered logit</a></li>
<li><a href="Ordered_probit" class="mw-redirect" title="Ordered probit">Ordered probit</a></li>
<li><a href="Poisson_regression" title="Poisson regression">Poisson</a></li></ul></td>
</tr><tr><td class="sidebar-content">
<ul><li><a href="Multilevel_model" title="Multilevel model">Multilevel model</a></li>
<li><a href="Fixed_effects_model" title="Fixed effects model">Fixed effects</a></li>
<li><a href="Random_effects_model" title="Random effects model">Random effects</a></li>
<li><a href="Mixed_model" title="Mixed model">Linear mixed-effects model</a></li>
<li><a href="Nonlinear_mixed-effects_model" title="Nonlinear mixed-effects model">Nonlinear mixed-effects model</a></li></ul></td>
</tr><tr><td class="sidebar-content">
<ul><li><a href="Nonlinear_regression" title="Nonlinear regression">Nonlinear regression</a></li>
<li><a href="Nonparametric_regression" title="Nonparametric regression">Nonparametric</a></li>
<li><a href="Semiparametric_regression" title="Semiparametric regression">Semiparametric</a></li>
<li><a href="Robust_regression" title="Robust regression">Robust</a></li>
<li><a href="Quantile_regression" title="Quantile regression">Quantile</a></li>
<li><a href="Isotonic_regression" title="Isotonic regression">Isotonic</a></li>
<li><a href="Principal_component_regression" title="Principal component regression">Principal components</a></li>
<li><a href="Least-angle_regression" title="Least-angle regression">Least angle</a></li>
<li><a href="Local_regression" title="Local regression">Local</a></li>
<li><a href="Segmented_regression" title="Segmented regression">Segmented</a></li></ul></td>
</tr><tr><td class="sidebar-content">
<ul><li><a href="Errors-in-variables_models" class="mw-redirect" title="Errors-in-variables models">Errors-in-variables</a></li></ul></td>
</tr><tr><th class="sidebar-heading">
Estimation</th></tr><tr><td class="sidebar-content">
<ul><li><a href="Least_squares" title="Least squares">Least squares</a></li>
<li><a href="Linear_least_squares" title="Linear least squares">Linear</a></li>
<li><a href="Non-linear_least_squares" title="Non-linear least squares">Non-linear</a></li></ul></td>
</tr><tr><td class="sidebar-content">
<ul><li><a href="Ordinary_least_squares" title="Ordinary least squares">Ordinary</a></li>
<li><a href="Weighted_least_squares" title="Weighted least squares">Weighted</a></li>
<li><a href="Generalized_least_squares" title="Generalized least squares">Generalized</a></li>
<li><a href="Generalized_estimating_equation" title="Generalized estimating equation">Generalized estimating equation</a></li></ul></td>
</tr><tr><td class="sidebar-content">
<ul><li><a href="Partial_least_squares_regression" title="Partial least squares regression">Partial</a></li>
<li><a href="Total_least_squares" title="Total least squares">Total</a></li>
<li><a href="Non-negative_least_squares" title="Non-negative least squares">Non-negative</a></li>
<li><a href="Tikhonov_regularization" class="mw-redirect" title="Tikhonov regularization">Ridge regression</a></li>
<li><a href="Regularized_least_squares" title="Regularized least squares">Regularized</a></li></ul></td>
</tr><tr><td class="sidebar-content">
<ul><li><a href="Least_absolute_deviations" title="Least absolute deviations">Least absolute deviations</a></li>
<li><a href="Iteratively_reweighted_least_squares" title="Iteratively reweighted least squares">Iteratively reweighted</a></li>
<li><a href="Bayesian_linear_regression" title="Bayesian linear regression">Bayesian</a></li>
<li><a href="Bayesian_multivariate_linear_regression" title="Bayesian multivariate linear regression">Bayesian multivariate</a></li>
<li><a href="Least-squares_spectral_analysis" title="Least-squares spectral analysis">Least-squares spectral analysis</a></li></ul></td>
</tr><tr><th class="sidebar-heading">
Background</th></tr><tr><td class="sidebar-content">
<ul><li><a href="Regression_validation" title="Regression validation">Regression validation</a></li>
<li><a href="Mean_and_predicted_response" class="mw-redirect" title="Mean and predicted response">Mean and predicted response</a></li>
<li><a href="Errors_and_residuals" title="Errors and residuals">Errors and residuals</a></li>
<li><a href="Goodness_of_fit" title="Goodness of fit">Goodness of fit</a></li>
<li><a href="Studentized_residual" title="Studentized residual">Studentized residual</a></li>
<li><a href="Gauss%E2%80%93Markov_theorem" title="Gauss–Markov theorem">Gauss–Markov theorem</a></li></ul></td>
</tr><tr><td class="sidebar-below">
<ul><li><span class="nowrap"><span class="skin-invert-image noviewer" typeof="mw:File"></span> </span><a href="Portal%3AMathematics" title="Portal:Mathematics">Mathematics portal</a></li></ul></td></tr><tr><td class="sidebar-navbar"><style data-mw-deduplicate="TemplateStyles:r1239400231">
/* start https://en.wikipedia.org/ */


.mw-parser-output .navbar{display:inline;font-size:88%;font-weight:normal}.mw-parser-output .navbar-collapse{float:left;text-align:left}.mw-parser-output .navbar-boxtext{word-spacing:0}.mw-parser-output .navbar ul{display:inline-block;white-space:nowrap;line-height:inherit}.mw-parser-output .navbar-brackets::before{margin-right:-0.125em;content:"[ "}.mw-parser-output .navbar-brackets::after{margin-left:-0.125em;content:" ]"}.mw-parser-output .navbar li{word-spacing:-0.125em}.mw-parser-output .navbar a>span,.mw-parser-output .navbar a>abbr{text-decoration:inherit}.mw-parser-output .navbar-mini abbr{font-variant:small-caps;border-bottom:none;text-decoration:none;cursor:inherit}.mw-parser-output .navbar-ct-full{font-size:114%;margin:0 7em}.mw-parser-output .navbar-ct-mini{font-size:114%;margin:0 4em}html.skin-theme-clientpref-night .mw-parser-output .navbar li a abbr{color:var(--color-base)!important}@media(prefers-color-scheme:dark){html.skin-theme-clientpref-os .mw-parser-output .navbar li a abbr{color:var(--color-base)!important}}@media print{.mw-parser-output .navbar{display:none!important}}


/* end https://en.wikipedia.org/ */
</style></td></tr></tbody></table>
<p>In <a href="Statistics" title="Statistics">statistics</a>, <b>multinomial logistic regression</b> is a <a href="Statistical_classification" title="Statistical classification">classification</a> method that generalizes <a href="Logistic_regression" title="Logistic regression">logistic regression</a> to <a href="Multiclass_classification" title="Multiclass classification">multiclass problems</a>, i.e. with more than two possible discrete outcomes.<sup id="cite_ref-1" class="reference"><a href="#cite_note-1"><span class="cite-bracket">[</span>1<span class="cite-bracket">]</span></a></sup> That is, it is a model that is used to predict the probabilities of the different possible outcomes of a <a href="Categorical_distribution" title="Categorical distribution">categorically distributed</a> <a href="Dependent_variable" class="mw-redirect" title="Dependent variable">dependent variable</a>, given a set of <a href="Independent_variable" class="mw-redirect" title="Independent variable">independent variables</a> (which may be real-valued, binary-valued, categorical-valued, etc.).
</p><p>Multinomial logistic regression is known by a variety of other names, including <b>polytomous LR</b>,<sup id="cite_ref-2" class="reference"><a href="#cite_note-2"><span class="cite-bracket">[</span>2<span class="cite-bracket">]</span></a></sup><sup id="cite_ref-3" class="reference"><a href="#cite_note-3"><span class="cite-bracket">[</span>3<span class="cite-bracket">]</span></a></sup> <b>multiclass LR</b>, <b><a href="Softmax_activation_function" class="mw-redirect" title="Softmax activation function">softmax</a> regression</b>, <b>multinomial logit</b> (<b>mlogit</b>), the <b>maximum entropy</b> (<b>MaxEnt</b>) classifier, and the <b>conditional maximum entropy model</b>.<sup id="cite_ref-malouf_4-0" class="reference"><a href="#cite_note-malouf-4"><span class="cite-bracket">[</span>4<span class="cite-bracket">]</span></a></sup>
</p>
<meta property="mw:PageProp/toc">
<div class="mw-heading mw-heading2"><h2 id="Background">Background</h2></div>
<p>Multinomial logistic regression is used when the <a href="Dependent_variable" class="mw-redirect" title="Dependent variable">dependent variable</a> in question is <a href="Level_of_measurement#Nominal_measurement" title="Level of measurement">nominal</a> (equivalently <i>categorical</i>, meaning that it falls into any one of a set of categories that cannot be ordered in any meaningful way) and for which there are more than two categories. Some examples would be:
</p>
<ul><li>Which major will a college student choose, given their grades, stated likes and dislikes, etc.?</li>
<li>Which blood type does a person have, given the results of various diagnostic tests?</li>
<li>In a hands-free mobile phone dialing application, which person's name was spoken, given various properties of the speech signal?</li>
<li>Which candidate will a person vote for, given particular demographic characteristics?</li>
<li>Which country will a firm locate an office in, given the characteristics of the firm and of the various candidate countries?</li></ul>
<p>These are all <a href="Statistical_classification" title="Statistical classification">statistical classification</a> problems. They all have in common a <a href="Dependent_variable" class="mw-redirect" title="Dependent variable">dependent variable</a> to be predicted that comes from one of a limited set of items that cannot be meaningfully ordered, as well as a set of <a href="Independent_variable" class="mw-redirect" title="Independent variable">independent variables</a> (also known as features, explanators, etc.), which are used to predict the dependent variable. Multinomial logistic regression is a particular solution to classification problems that use a linear combination of the observed features and some problem-specific parameters to estimate the probability of each particular value of the dependent variable. The best values of the parameters for a given problem are usually determined from some training data (e.g. some people for whom both the diagnostic test results and blood types are known, or some examples of known words being spoken).
</p>
<div class="mw-heading mw-heading2"><h2 id="Assumptions">Assumptions</h2></div>
<p>The multinomial logistic model assumes that data are case-specific; that is, each independent variable has a single value for each case. As with other types of regression, there is no need for the independent variables to be <a href="Statistically_independent" class="mw-redirect" title="Statistically independent">statistically independent</a> from each other (unlike, for example, in a <a href="Naive_Bayes_classifier" title="Naive Bayes classifier">naive Bayes classifier</a>); however, <a href="Multicollinearity" title="Multicollinearity">collinearity</a> is assumed to be relatively low, as it becomes difficult to differentiate between the impact of several variables if this is not the case.<sup id="cite_ref-5" class="reference"><a href="#cite_note-5"><span class="cite-bracket">[</span>5<span class="cite-bracket">]</span></a></sup>
</p><p>If the multinomial logit is used to model choices, it relies on the assumption of <a href="Independence_of_irrelevant_alternatives" title="Independence of irrelevant alternatives">independence of irrelevant alternatives</a> (IIA), which is not always desirable. This assumption states that the odds of preferring one class over another do not depend on the presence or absence of other "irrelevant" alternatives. For example, the relative probabilities of taking a car or bus to work do not change if a bicycle is added as an additional possibility. This allows the choice of <i>K</i> alternatives to be modeled as a set of <i>K</i>&nbsp;−&nbsp;1 independent binary choices, in which one alternative is chosen as a "pivot" and the other <i>K</i>&nbsp;−&nbsp;1 compared against it, one at a time. The IIA hypothesis is a core hypothesis in rational choice theory; however numerous studies in psychology show that individuals often violate this assumption when making choices. An example of a problem case arises if choices include a car and a blue bus. Suppose the odds ratio between the two is 1&nbsp;: 1. Now if the option of a red bus is introduced, a person may be indifferent between a red and a blue bus, and hence may exhibit a car&nbsp;: blue bus&nbsp;: red bus odds ratio of 1&nbsp;: 0.5&nbsp;: 0.5, thus maintaining a 1&nbsp;: 1 ratio of car&nbsp;: any bus while adopting a changed car&nbsp;: blue bus ratio of 1&nbsp;: 0.5. Here the red bus option was not in fact irrelevant, because a red bus was a <a href="Perfect_substitute" class="mw-redirect" title="Perfect substitute">perfect substitute</a> for a blue bus.
</p><p>If the multinomial logit is used to model choices, it may in some situations impose too much constraint on the relative preferences between the different alternatives. It is especially important to take into account if the analysis aims to predict how choices would change if one alternative were to disappear (for instance if one political candidate withdraws from a three candidate race). Other models like the <a href="Nested_logit" class="mw-redirect" title="Nested logit">nested logit</a> or the <a href="Multinomial_probit" title="Multinomial probit">multinomial probit</a> may be used in such cases as they allow for violation of the IIA.<sup id="cite_ref-6" class="reference"><a href="#cite_note-6"><span class="cite-bracket">[</span>6<span class="cite-bracket">]</span></a></sup>
</p>
<div class="mw-heading mw-heading2"><h2 id="Model">Model</h2></div>
<div role="note" class="hatnote navigation-not-searchable">See also: <a href="Logistic_regression" title="Logistic regression">Logistic regression</a></div>
<div class="mw-heading mw-heading3"><h3 id="Introduction">Introduction</h3></div>
<p>There are multiple equivalent ways to describe the mathematical model underlying multinomial logistic regression. This can make it difficult to compare different treatments of the subject in different texts. The article on <a href="Logistic_regression" title="Logistic regression">logistic regression</a> presents a number of equivalent formulations of simple logistic regression, and many of these have analogues in the multinomial logit model.
</p><p>The idea behind all of them, as in many other <a href="Statistical_classification" title="Statistical classification">statistical classification</a> techniques, is to construct a <a href="Linear_predictor_function" title="Linear predictor function">linear predictor function</a> that constructs a score from a set of weights that are <a href="Linear_combination" title="Linear combination">linearly combined</a> with the explanatory variables (features) of a given observation using a <a href="Dot_product" title="Dot product">dot product</a>:
</p>
<dl><dd><span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle \operatorname {score} (\mathbf {X} _{i},k)={\boldsymbol {\beta }}_{k}\cdot \mathbf {X} _{i},}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>score</mi>
<mo>⁡<!-- ⁡ --></mo>
<mo stretchy="false">(</mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold">X</mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
<mo>,</mo>
<mi>k</mi>
<mo stretchy="false">)</mo>
<mo>=</mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold-italic">β<!-- β --></mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>k</mi>
</mrow>
</msub>
<mo>⋅<!-- ⋅ --></mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold">X</mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
<mo>,</mo>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle \operatorname {score} (\mathbf {X} _{i},k)={\boldsymbol {\beta }}_{k}\cdot \mathbf {X} _{i},}</annotation>
</semantics>
</math></span><img src="./d7810d95ad9b2bfbe7384432e83fa311ace76b03.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.838ex; width:22.795ex; height:2.843ex;" alt="{\displaystyle \operatorname {score} (\mathbf {X} _{i},k)={\boldsymbol {\beta }}_{k}\cdot \mathbf {X} _{i},}" loading="lazy"></span></dd></dl>
<p>where <b>X</b><sub><i>i</i></sub> is the vector of explanatory variables describing observation <i>i</i>, <b>β</b><sub><i>k</i></sub> is a vector of weights (or <a href="Regression_coefficient" class="mw-redirect" title="Regression coefficient">regression coefficients</a>) corresponding to outcome <i>k</i>, and score(<b>X</b><sub><i>i</i></sub>, <i>k</i>) is the score associated with assigning observation <i>i</i> to category <i>k</i>. In <a href="Discrete_choice" title="Discrete choice">discrete choice</a> theory, where observations represent people and outcomes represent choices, the score is considered the <a href="Utility" title="Utility">utility</a> associated with person <i>i</i> choosing outcome <i>k</i>. The predicted outcome is the one with the highest score.
</p><p>The difference between the multinomial logit model and numerous other methods, models, algorithms, etc. with the same basic setup (the <a href="Perceptron" title="Perceptron">perceptron</a> algorithm, <a href="Support_vector_machine" title="Support vector machine">support vector machines</a>, <a href="Linear_discriminant_analysis" title="Linear discriminant analysis">linear discriminant analysis</a>, etc.) is the procedure for determining (training) the optimal weights/coefficients and the way that the score is interpreted. In particular, in the multinomial logit model, the score can directly be converted to a probability value, indicating the <a href="Probability" title="Probability">probability</a> of observation <i>i</i> choosing outcome <i>k</i> given the measured characteristics of the observation. This provides a principled way of incorporating the prediction of a particular multinomial logit model into a larger procedure that may involve multiple such predictions, each with a possibility of error. Without such means of combining predictions, errors tend to multiply. For example, imagine a large <a href="Predictive_modelling" title="Predictive modelling">predictive model</a> that is broken down into a series of submodels where the prediction of a given submodel is used as the input of another submodel, and that prediction is in turn used as the input into a third submodel, etc. If each submodel has 90% accuracy in its predictions, and there are five submodels in series, then the overall model has only 0.9<sup>5</sup> = 59% accuracy. If each submodel has 80% accuracy, then overall accuracy drops to 0.8<sup>5</sup> = 33% accuracy. This issue is known as <a href="Error_propagation" class="mw-redirect" title="Error propagation">error propagation</a> and is a serious problem in real-world predictive models, which are usually composed of numerous parts. Predicting probabilities of each possible outcome, rather than simply making a single optimal prediction, is one means of alleviating this issue.
</p>
<div class="mw-heading mw-heading3"><h3 id="Setup">Setup</h3></div>
<p>The basic setup is the same as in <a href="Logistic_regression" title="Logistic regression">logistic regression</a>, the only difference being that the <a href="Dependent_variable" class="mw-redirect" title="Dependent variable">dependent variables</a> are <a href="Categorical_variable" title="Categorical variable">categorical</a> rather than <a href="Binary_variable" class="mw-redirect" title="Binary variable">binary</a>, i.e. there are <i>K</i> possible outcomes rather than just two. The following description is somewhat shortened; for more details, consult the <a href="Logistic_regression" title="Logistic regression">logistic regression</a> article.
</p>
<div class="mw-heading mw-heading4"><h4 id="Data_points">Data points</h4></div>
<p>Specifically, it is assumed that we have a series of <i>N</i> observed data points. Each data point <i>i</i> (ranging from 1 to <i>N</i>) consists of a set of <i>M</i> explanatory variables <i>x</i><sub>1,<i>i</i></sub> ... <i>x</i><sub><i>M,i</i></sub> (also known as <a href="Independent_variable" class="mw-redirect" title="Independent variable">independent variables</a>, predictor variables, features, etc.), and an associated <a href="Categorical_variable" title="Categorical variable">categorical</a> outcome <i>Y</i><sub><i>i</i></sub> (also known as <a href="Dependent_variable" class="mw-redirect" title="Dependent variable">dependent variable</a>, response variable), which can take on one of <i>K</i> possible values. These possible values represent logically separate categories (e.g. different political parties, blood types, etc.), and are often described mathematically by arbitrarily assigning each a number from 1 to <i>K</i>. The explanatory variables and outcome represent observed properties of the data points, and are often thought of as originating in the observations of <i>N</i> "experiments" — although an "experiment" may consist of nothing more than gathering data. The goal of multinomial logistic regression is to construct a model that explains the relationship between the explanatory variables and the outcome, so that the outcome of a new "experiment" can be correctly predicted for a new data point for which the explanatory variables, but not the outcome, are available. In the process, the model attempts to explain the relative effect of differing explanatory variables on the outcome.
</p><p>Some examples:
</p>
<ul><li>The observed outcomes are different variants of a disease such as <a href="Hepatitis" title="Hepatitis">hepatitis</a> (possibly including "no disease" and/or other related diseases) in a set of patients, and the explanatory variables might be characteristics of the patients thought to be pertinent (sex, race, age, <a href="Blood_pressure" title="Blood pressure">blood pressure</a>, outcomes of various liver-function tests, etc.). The goal is then to predict which disease is causing the observed liver-related symptoms in a new patient.</li>
<li>The observed outcomes are the party chosen by a set of people in an election, and the explanatory variables are the demographic characteristics of each person (e.g. sex, race, age, income, etc.). The goal is then to predict the likely vote of a new voter with given characteristics.</li></ul>
<div class="mw-heading mw-heading4"><h4 id="Linear_predictor">Linear predictor</h4></div>
<p>As in other forms of linear regression, multinomial logistic regression uses a <a href="Linear_predictor_function" title="Linear predictor function">linear predictor function</a> <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle f(k,i)}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>f</mi>
<mo stretchy="false">(</mo>
<mi>k</mi>
<mo>,</mo>
<mi>i</mi>
<mo stretchy="false">)</mo>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle f(k,i)}</annotation>
</semantics>
</math></span><img src="./811817b0a554c4823cf74b54ec0a28fbbcc44756.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.838ex; width:6.136ex; height:2.843ex;" alt="{\displaystyle f(k,i)}" loading="lazy"></span> to predict the probability that observation <i>i</i> has outcome <i>k</i>, of the following form:
</p>
<dl><dd><span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle f(k,i)=\beta _{0,k}+\beta _{1,k}x_{1,i}+\beta _{2,k}x_{2,i}+\cdots +\beta _{M,k}x_{M,i},}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>f</mi>
<mo stretchy="false">(</mo>
<mi>k</mi>
<mo>,</mo>
<mi>i</mi>
<mo stretchy="false">)</mo>
<mo>=</mo>
<msub>
<mi>β<!-- β --></mi>
<mrow class="MJX-TeXAtom-ORD">
<mn>0</mn>
<mo>,</mo>
<mi>k</mi>
</mrow>
</msub>
<mo>+</mo>
<msub>
<mi>β<!-- β --></mi>
<mrow class="MJX-TeXAtom-ORD">
<mn>1</mn>
<mo>,</mo>
<mi>k</mi>
</mrow>
</msub>
<msub>
<mi>x</mi>
<mrow class="MJX-TeXAtom-ORD">
<mn>1</mn>
<mo>,</mo>
<mi>i</mi>
</mrow>
</msub>
<mo>+</mo>
<msub>
<mi>β<!-- β --></mi>
<mrow class="MJX-TeXAtom-ORD">
<mn>2</mn>
<mo>,</mo>
<mi>k</mi>
</mrow>
</msub>
<msub>
<mi>x</mi>
<mrow class="MJX-TeXAtom-ORD">
<mn>2</mn>
<mo>,</mo>
<mi>i</mi>
</mrow>
</msub>
<mo>+</mo>
<mo>⋯<!-- ⋯ --></mo>
<mo>+</mo>
<msub>
<mi>β<!-- β --></mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>M</mi>
<mo>,</mo>
<mi>k</mi>
</mrow>
</msub>
<msub>
<mi>x</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>M</mi>
<mo>,</mo>
<mi>i</mi>
</mrow>
</msub>
<mo>,</mo>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle f(k,i)=\beta _{0,k}+\beta _{1,k}x_{1,i}+\beta _{2,k}x_{2,i}+\cdots +\beta _{M,k}x_{M,i},}</annotation>
</semantics>
</math></span><img src="./2cb74185a381069c19dfc9ceb09c34b0a887ac8a.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -1.005ex; width:50.737ex; height:3.009ex;" alt="{\displaystyle f(k,i)=\beta _{0,k}+\beta _{1,k}x_{1,i}+\beta _{2,k}x_{2,i}+\cdots +\beta _{M,k}x_{M,i},}" loading="lazy"></span></dd></dl>
<p>where <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle \beta _{m,k}}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<msub>
<mi>β<!-- β --></mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>m</mi>
<mo>,</mo>
<mi>k</mi>
</mrow>
</msub>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle \beta _{m,k}}</annotation>
</semantics>
</math></span><img src="./b4dcbd15fb2f28d1babf127afabe8ea103e8b457.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -1.005ex; width:4.305ex; height:2.843ex;" alt="{\displaystyle \beta _{m,k}}" loading="lazy"></span> is a <a href="Regression_coefficient" class="mw-redirect" title="Regression coefficient">regression coefficient</a> associated with the <i>m</i>th explanatory variable and the <i>k</i>th outcome. As explained in the <a href="Logistic_regression" title="Logistic regression">logistic regression</a> article, the regression coefficients and explanatory variables are normally grouped into vectors of size <i>M</i>&nbsp;+&nbsp;1, so that the predictor function can be written more compactly:
</p>
<dl><dd><span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle f(k,i)={\boldsymbol {\beta }}_{k}\cdot \mathbf {x} _{i},}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>f</mi>
<mo stretchy="false">(</mo>
<mi>k</mi>
<mo>,</mo>
<mi>i</mi>
<mo stretchy="false">)</mo>
<mo>=</mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold-italic">β<!-- β --></mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>k</mi>
</mrow>
</msub>
<mo>⋅<!-- ⋅ --></mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold">x</mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
<mo>,</mo>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle f(k,i)={\boldsymbol {\beta }}_{k}\cdot \mathbf {x} _{i},}</annotation>
</semantics>
</math></span><img src="./eb11a10791be260ba57f1e213e4d6afc79390674.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.838ex; width:16.393ex; height:2.843ex;" alt="{\displaystyle f(k,i)={\boldsymbol {\beta }}_{k}\cdot \mathbf {x} _{i},}" loading="lazy"></span></dd></dl>
<p>where <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle {\boldsymbol {\beta }}_{k}}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold-italic">β<!-- β --></mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>k</mi>
</mrow>
</msub>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle {\boldsymbol {\beta }}_{k}}</annotation>
</semantics>
</math></span><img src="./07f1bcb6340535f54accaf88124f4cd5ba1d4fc8.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.838ex; width:2.623ex; height:2.676ex;" alt="{\displaystyle {\boldsymbol {\beta }}_{k}}" loading="lazy"></span> is the set of regression coefficients associated with outcome <i>k</i>, and <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle \mathbf {x} _{i}}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold">x</mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle \mathbf {x} _{i}}</annotation>
</semantics>
</math></span><img src="./57d2ef3df60acdb53bdf90535264041fea7231cd.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.671ex; width:2.211ex; height:2.009ex;" alt="{\displaystyle \mathbf {x} _{i}}" loading="lazy"></span> (a row vector) is the set of explanatory variables associated with observation <i>i</i>, prepended by a 1 in entry 0.
</p>
<div class="mw-heading mw-heading3"><h3 id="As_a_set_of_independent_binary_regressions">As a set of independent binary regressions</h3></div>
<p>To arrive at the multinomial logit model, one can imagine, for <i>K</i> possible outcomes, running <i>K</i> independent binary logistic regression models, in which one outcome is chosen as a "pivot" and then the other <i>K</i>&nbsp;−&nbsp;1 outcomes are separately regressed against the pivot outcome. If outcome <i>K</i> (the last outcome) is chosen as the pivot, the <i>K</i>&nbsp;−&nbsp;1 regression equations are:
</p>
<dl><dd><span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle \ln {\frac {\Pr(Y_{i}=k)}{\Pr(Y_{i}=K)}}\,=\,{\boldsymbol {\beta }}_{k}\cdot \mathbf {X} _{i},\;\;\;\;\;\;1\leq k<K}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>ln</mi>
<mo>⁡<!-- ⁡ --></mo>
<mrow class="MJX-TeXAtom-ORD">
<mfrac>
<mrow>
<mo movablelimits="true" form="prefix">Pr</mo>
<mo stretchy="false">(</mo>
<msub>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
<mo>=</mo>
<mi>k</mi>
<mo stretchy="false">)</mo>
</mrow>
<mrow>
<mo movablelimits="true" form="prefix">Pr</mo>
<mo stretchy="false">(</mo>
<msub>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
<mo>=</mo>
<mi>K</mi>
<mo stretchy="false">)</mo>
</mrow>
</mfrac>
</mrow>
<mspace width="thinmathspace"></mspace>
<mo>=</mo>
<mspace width="thinmathspace"></mspace>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold-italic">β<!-- β --></mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>k</mi>
</mrow>
</msub>
<mo>⋅<!-- ⋅ --></mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold">X</mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
<mo>,</mo>
<mspace width="thickmathspace"></mspace>
<mspace width="thickmathspace"></mspace>
<mspace width="thickmathspace"></mspace>
<mspace width="thickmathspace"></mspace>
<mspace width="thickmathspace"></mspace>
<mspace width="thickmathspace"></mspace>
<mn>1</mn>
<mo>≤<!-- ≤ --></mo>
<mi>k</mi>
<mo>&lt;</mo>
<mi>K</mi>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle \ln {\frac {\Pr(Y_{i}=k)}{\Pr(Y_{i}=K)}}\,=\,{\boldsymbol {\beta }}_{k}\cdot \mathbf {X} _{i},\;\;\;\;\;\;1\leq k&lt;K}</annotation>
</semantics>
</math></span><img src="./70cdde1873f5fbe834ea45e83d168e54c12192d0.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -2.671ex; width:41.316ex; height:6.509ex;" alt="{\displaystyle \ln {\frac {\Pr(Y_{i}=k)}{\Pr(Y_{i}=K)}}\,=\,{\boldsymbol {\beta }}_{k}\cdot \mathbf {X} _{i},\;\;\;\;\;\;1\leq k<K}" loading="lazy"></span>.</dd></dl>
<p>This formulation is also known as the <a href="Compositional_data#Additive_log_ratio_transform" title="Compositional data">Additive Log Ratio</a> transform commonly used in compositional data analysis. In other applications it’s referred to as “relative risk”.<sup id="cite_ref-7" class="reference"><a href="#cite_note-7"><span class="cite-bracket">[</span>7<span class="cite-bracket">]</span></a></sup>
</p><p>If we exponentiate both sides and solve for the probabilities, we get:
</p>
<dl><dd><span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle \Pr(Y_{i}=k)\,=\,{\Pr(Y_{i}=K)}\;e^{{\boldsymbol {\beta }}_{k}\cdot \mathbf {X} _{i}},\;\;\;\;\;\;1\leq k<K}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mo movablelimits="true" form="prefix">Pr</mo>
<mo stretchy="false">(</mo>
<msub>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
<mo>=</mo>
<mi>k</mi>
<mo stretchy="false">)</mo>
<mspace width="thinmathspace"></mspace>
<mo>=</mo>
<mspace width="thinmathspace"></mspace>
<mrow class="MJX-TeXAtom-ORD">
<mo movablelimits="true" form="prefix">Pr</mo>
<mo stretchy="false">(</mo>
<msub>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
<mo>=</mo>
<mi>K</mi>
<mo stretchy="false">)</mo>
</mrow>
<mspace width="thickmathspace"></mspace>
<msup>
<mi>e</mi>
<mrow class="MJX-TeXAtom-ORD">
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold-italic">β<!-- β --></mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>k</mi>
</mrow>
</msub>
<mo>⋅<!-- ⋅ --></mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold">X</mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
</mrow>
</msup>
<mo>,</mo>
<mspace width="thickmathspace"></mspace>
<mspace width="thickmathspace"></mspace>
<mspace width="thickmathspace"></mspace>
<mspace width="thickmathspace"></mspace>
<mspace width="thickmathspace"></mspace>
<mspace width="thickmathspace"></mspace>
<mn>1</mn>
<mo>≤<!-- ≤ --></mo>
<mi>k</mi>
<mo>&lt;</mo>
<mi>K</mi>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle \Pr(Y_{i}=k)\,=\,{\Pr(Y_{i}=K)}\;e^{{\boldsymbol {\beta }}_{k}\cdot \mathbf {X} _{i}},\;\;\;\;\;\;1\leq k&lt;K}</annotation>
</semantics>
</math></span><img src="./b24ad33ccd1649cc06bfa38c6e4350e79c25214d.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.838ex; width:48.212ex; height:3.176ex;" alt="{\displaystyle \Pr(Y_{i}=k)\,=\,{\Pr(Y_{i}=K)}\;e^{{\boldsymbol {\beta }}_{k}\cdot \mathbf {X} _{i}},\;\;\;\;\;\;1\leq k<K}" loading="lazy"></span></dd></dl>
<p>Using the fact that all <i>K</i> of the probabilities must sum to one, we find:
</p>
<dl><dd><span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle {\begin{aligned}\Pr(Y_{i}=K)={}&amp;1-\sum _{j=1}^{K-1}\Pr(Y_{i}=j)\\={}&amp;1-\sum _{j=1}^{K-1}{\Pr(Y_{i}=K)}\;e^{{\boldsymbol {\beta }}_{j}\cdot \mathbf {X} _{i}}\;\;\Rightarrow \;\;\Pr(Y_{i}=K)\\={}&amp;{\frac {1}{1+\sum _{j=1}^{K-1}e^{{\boldsymbol {\beta }}_{j}\cdot \mathbf {X} _{i}}}}.\end{aligned}}}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mrow class="MJX-TeXAtom-ORD">
<mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true">
<mtr>
<mtd>
<mo movablelimits="true" form="prefix">Pr</mo>
<mo stretchy="false">(</mo>
<msub>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
<mo>=</mo>
<mi>K</mi>
<mo stretchy="false">)</mo>
<mo>=</mo>
<mrow class="MJX-TeXAtom-ORD">

</mrow>
</mtd>
<mtd>
<mn>1</mn>
<mo>−<!-- − --></mo>
<munderover>
<mo>∑<!-- ∑ --></mo>
<mrow class="MJX-TeXAtom-ORD">
<mi>j</mi>
<mo>=</mo>
<mn>1</mn>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>K</mi>
<mo>−<!-- − --></mo>
<mn>1</mn>
</mrow>
</munderover>
<mo movablelimits="true" form="prefix">Pr</mo>
<mo stretchy="false">(</mo>
<msub>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
<mo>=</mo>
<mi>j</mi>
<mo stretchy="false">)</mo>
</mtd>
</mtr>
<mtr>
<mtd>
<mo>=</mo>
<mrow class="MJX-TeXAtom-ORD">

</mrow>
</mtd>
<mtd>
<mn>1</mn>
<mo>−<!-- − --></mo>
<munderover>
<mo>∑<!-- ∑ --></mo>
<mrow class="MJX-TeXAtom-ORD">
<mi>j</mi>
<mo>=</mo>
<mn>1</mn>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>K</mi>
<mo>−<!-- − --></mo>
<mn>1</mn>
</mrow>
</munderover>
<mrow class="MJX-TeXAtom-ORD">
<mo movablelimits="true" form="prefix">Pr</mo>
<mo stretchy="false">(</mo>
<msub>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
<mo>=</mo>
<mi>K</mi>
<mo stretchy="false">)</mo>
</mrow>
<mspace width="thickmathspace"></mspace>
<msup>
<mi>e</mi>
<mrow class="MJX-TeXAtom-ORD">
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold-italic">β<!-- β --></mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>j</mi>
</mrow>
</msub>
<mo>⋅<!-- ⋅ --></mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold">X</mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
</mrow>
</msup>
<mspace width="thickmathspace"></mspace>
<mspace width="thickmathspace"></mspace>
<mo stretchy="false">⇒<!-- ⇒ --></mo>
<mspace width="thickmathspace"></mspace>
<mspace width="thickmathspace"></mspace>
<mo movablelimits="true" form="prefix">Pr</mo>
<mo stretchy="false">(</mo>
<msub>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
<mo>=</mo>
<mi>K</mi>
<mo stretchy="false">)</mo>
</mtd>
</mtr>
<mtr>
<mtd>
<mo>=</mo>
<mrow class="MJX-TeXAtom-ORD">

</mrow>
</mtd>
<mtd>
<mrow class="MJX-TeXAtom-ORD">
<mfrac>
<mn>1</mn>
<mrow>
<mn>1</mn>
<mo>+</mo>
<munderover>
<mo>∑<!-- ∑ --></mo>
<mrow class="MJX-TeXAtom-ORD">
<mi>j</mi>
<mo>=</mo>
<mn>1</mn>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>K</mi>
<mo>−<!-- − --></mo>
<mn>1</mn>
</mrow>
</munderover>
<msup>
<mi>e</mi>
<mrow class="MJX-TeXAtom-ORD">
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold-italic">β<!-- β --></mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>j</mi>
</mrow>
</msub>
<mo>⋅<!-- ⋅ --></mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold">X</mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
</mrow>
</msup>
</mrow>
</mfrac>
</mrow>
<mo>.</mo>
</mtd>
</mtr>
</mtable>
</mrow>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle {\begin{aligned}\Pr(Y_{i}=K)={}&amp;1-\sum _{j=1}^{K-1}\Pr(Y_{i}=j)\\={}&amp;1-\sum _{j=1}^{K-1}{\Pr(Y_{i}=K)}\;e^{{\boldsymbol {\beta }}_{j}\cdot \mathbf {X} _{i}}\;\;\Rightarrow \;\;\Pr(Y_{i}=K)\\={}&amp;{\frac {1}{1+\sum _{j=1}^{K-1}e^{{\boldsymbol {\beta }}_{j}\cdot \mathbf {X} _{i}}}}.\end{aligned}}}</annotation>
</semantics>
</math></span><img src="./d9d1b5b239dafc937d2f18145d1b2adf30c5039d.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -10.671ex; width:59.121ex; height:22.509ex;" alt="{\displaystyle {\begin{aligned}\Pr(Y_{i}=K)={}&amp;1-\sum _{j=1}^{K-1}\Pr(Y_{i}=j)\\={}&amp;1-\sum _{j=1}^{K-1}{\Pr(Y_{i}=K)}\;e^{{\boldsymbol {\beta }}_{j}\cdot \mathbf {X} _{i}}\;\;\Rightarrow \;\;\Pr(Y_{i}=K)\\={}&amp;{\frac {1}{1+\sum _{j=1}^{K-1}e^{{\boldsymbol {\beta }}_{j}\cdot \mathbf {X} _{i}}}}.\end{aligned}}}" loading="lazy"></span></dd></dl>
<p>We can use this to find the other probabilities:
</p>
<dl><dd><span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle \Pr(Y_{i}=k)={\frac {e^{{\boldsymbol {\beta }}_{k}\cdot \mathbf {X} _{i}}}{1+\sum _{j=1}^{K-1}e^{{\boldsymbol {\beta }}_{j}\cdot \mathbf {X} _{i}}}},\;\;\;\;\;\;1\leq k<K}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mo movablelimits="true" form="prefix">Pr</mo>
<mo stretchy="false">(</mo>
<msub>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
<mo>=</mo>
<mi>k</mi>
<mo stretchy="false">)</mo>
<mo>=</mo>
<mrow class="MJX-TeXAtom-ORD">
<mfrac>
<msup>
<mi>e</mi>
<mrow class="MJX-TeXAtom-ORD">
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold-italic">β<!-- β --></mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>k</mi>
</mrow>
</msub>
<mo>⋅<!-- ⋅ --></mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold">X</mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
</mrow>
</msup>
<mrow>
<mn>1</mn>
<mo>+</mo>
<munderover>
<mo>∑<!-- ∑ --></mo>
<mrow class="MJX-TeXAtom-ORD">
<mi>j</mi>
<mo>=</mo>
<mn>1</mn>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>K</mi>
<mo>−<!-- − --></mo>
<mn>1</mn>
</mrow>
</munderover>
<msup>
<mi>e</mi>
<mrow class="MJX-TeXAtom-ORD">
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold-italic">β<!-- β --></mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>j</mi>
</mrow>
</msub>
<mo>⋅<!-- ⋅ --></mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold">X</mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
</mrow>
</msup>
</mrow>
</mfrac>
</mrow>
<mo>,</mo>
<mspace width="thickmathspace"></mspace>
<mspace width="thickmathspace"></mspace>
<mspace width="thickmathspace"></mspace>
<mspace width="thickmathspace"></mspace>
<mspace width="thickmathspace"></mspace>
<mspace width="thickmathspace"></mspace>
<mn>1</mn>
<mo>≤<!-- ≤ --></mo>
<mi>k</mi>
<mo>&lt;</mo>
<mi>K</mi>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle \Pr(Y_{i}=k)={\frac {e^{{\boldsymbol {\beta }}_{k}\cdot \mathbf {X} _{i}}}{1+\sum _{j=1}^{K-1}e^{{\boldsymbol {\beta }}_{j}\cdot \mathbf {X} _{i}}}},\;\;\;\;\;\;1\leq k&lt;K}</annotation>
</semantics>
</math></span><img src="./a468beb77dc39e5b212d930a4db673fc0ffd1c25.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -3.505ex; width:46.502ex; height:7.343ex;" alt="{\displaystyle \Pr(Y_{i}=k)={\frac {e^{{\boldsymbol {\beta }}_{k}\cdot \mathbf {X} _{i}}}{1+\sum _{j=1}^{K-1}e^{{\boldsymbol {\beta }}_{j}\cdot \mathbf {X} _{i}}}},\;\;\;\;\;\;1\leq k<K}" loading="lazy"></span>.</dd></dl>
<p>The fact that we run multiple regressions reveals why the model relies on the assumption of <a href="Independence_of_irrelevant_alternatives" title="Independence of irrelevant alternatives">independence of irrelevant alternatives</a> described above.
</p>
<div class="mw-heading mw-heading3"><h3 id="Estimating_the_coefficients">Estimating the coefficients</h3></div>
<p>The unknown parameters in each vector <i><b>β</b><sub>k</sub></i> are typically jointly estimated by <a href="Maximum_a_posteriori" class="mw-redirect" title="Maximum a posteriori">maximum a posteriori</a> (MAP) estimation, which is an extension of <a href="Maximum_likelihood" class="mw-redirect" title="Maximum likelihood">maximum likelihood</a> using <a href="Regularization_(mathematics)" title="Regularization (mathematics)">regularization</a> of the weights to prevent pathological solutions (usually a squared regularizing function, which is equivalent to placing a zero-mean <a href="Gaussian_distribution" class="mw-redirect" title="Gaussian distribution">Gaussian</a> <a href="Prior_distribution" class="mw-redirect" title="Prior distribution">prior distribution</a> on the weights, but other distributions are also possible). The solution is typically found using an iterative procedure such as <a href="Generalized_iterative_scaling" title="Generalized iterative scaling">generalized iterative scaling</a>,<sup id="cite_ref-8" class="reference"><a href="#cite_note-8"><span class="cite-bracket">[</span>8<span class="cite-bracket">]</span></a></sup> <a href="Iteratively_reweighted_least_squares" title="Iteratively reweighted least squares">iteratively reweighted least squares</a> (IRLS),<sup id="cite_ref-9" class="reference"><a href="#cite_note-9"><span class="cite-bracket">[</span>9<span class="cite-bracket">]</span></a></sup> by means of <a href="Gradient-based_optimization" class="mw-redirect" title="Gradient-based optimization">gradient-based optimization</a> algorithms such as <a href="L-BFGS" class="mw-redirect" title="L-BFGS">L-BFGS</a>,<sup id="cite_ref-malouf_4-1" class="reference"><a href="#cite_note-malouf-4"><span class="cite-bracket">[</span>4<span class="cite-bracket">]</span></a></sup> or by specialized <a href="Coordinate_descent" title="Coordinate descent">coordinate descent</a> algorithms.<sup id="cite_ref-10" class="reference"><a href="#cite_note-10"><span class="cite-bracket">[</span>10<span class="cite-bracket">]</span></a></sup>
</p>
<div class="mw-heading mw-heading3"><h3 id="As_a_log-linear_model">As a log-linear model</h3></div>
<p>The formulation of binary logistic regression as a <a href="Logistic_regression#log-linear_model" title="Logistic regression">log-linear model</a> can be directly extended to multi-way regression. That is, we model the <a href="Logarithm" title="Logarithm">logarithm</a> of the probability of seeing a given output using the linear predictor as well as an additional <a href="Normalization_factor" class="mw-redirect" title="Normalization factor">normalization factor</a>, the logarithm of the <a href="Partition_function_(mathematics)" title="Partition function (mathematics)">partition function</a>:
</p>
<dl><dd><span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle \ln \Pr(Y_{i}=k)={\boldsymbol {\beta }}_{k}\cdot \mathbf {X} _{i}-\ln Z,\;\;\;\;\;\;1\leq k\leq K.}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>ln</mi>
<mo>⁡<!-- ⁡ --></mo>
<mo movablelimits="true" form="prefix">Pr</mo>
<mo stretchy="false">(</mo>
<msub>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
<mo>=</mo>
<mi>k</mi>
<mo stretchy="false">)</mo>
<mo>=</mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold-italic">β<!-- β --></mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>k</mi>
</mrow>
</msub>
<mo>⋅<!-- ⋅ --></mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold">X</mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
<mo>−<!-- − --></mo>
<mi>ln</mi>
<mo>⁡<!-- ⁡ --></mo>
<mi>Z</mi>
<mo>,</mo>
<mspace width="thickmathspace"></mspace>
<mspace width="thickmathspace"></mspace>
<mspace width="thickmathspace"></mspace>
<mspace width="thickmathspace"></mspace>
<mspace width="thickmathspace"></mspace>
<mspace width="thickmathspace"></mspace>
<mn>1</mn>
<mo>≤<!-- ≤ --></mo>
<mi>k</mi>
<mo>≤<!-- ≤ --></mo>
<mi>K</mi>
<mo>.</mo>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle \ln \Pr(Y_{i}=k)={\boldsymbol {\beta }}_{k}\cdot \mathbf {X} _{i}-\ln Z,\;\;\;\;\;\;1\leq k\leq K.}</annotation>
</semantics>
</math></span><img src="./4c63052d00b4b5350b1d9e1ec37c8a47891046ce.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.838ex; width:46.345ex; height:2.843ex;" alt="{\displaystyle \ln \Pr(Y_{i}=k)={\boldsymbol {\beta }}_{k}\cdot \mathbf {X} _{i}-\ln Z,\;\;\;\;\;\;1\leq k\leq K.}" loading="lazy"></span></dd></dl>
<p>As in the binary case, we need an extra term <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle -\ln Z}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mo>−<!-- − --></mo>
<mi>ln</mi>
<mo>⁡<!-- ⁡ --></mo>
<mi>Z</mi>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle -\ln Z}</annotation>
</semantics>
</math></span><img src="./bfc4529e0b901532d643b75ee89a2235538218ed.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.505ex; width:6.202ex; height:2.343ex;" alt="{\displaystyle -\ln Z}" loading="lazy"></span> to ensure that the whole set of probabilities forms a <a href="Probability_distribution" title="Probability distribution">probability distribution</a>, i.e. so that they all sum to one:
</p>
<dl><dd><span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle \sum _{k=1}^{K}\Pr(Y_{i}=k)=1}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<munderover>
<mo>∑<!-- ∑ --></mo>
<mrow class="MJX-TeXAtom-ORD">
<mi>k</mi>
<mo>=</mo>
<mn>1</mn>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>K</mi>
</mrow>
</munderover>
<mo movablelimits="true" form="prefix">Pr</mo>
<mo stretchy="false">(</mo>
<msub>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
<mo>=</mo>
<mi>k</mi>
<mo stretchy="false">)</mo>
<mo>=</mo>
<mn>1</mn>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle \sum _{k=1}^{K}\Pr(Y_{i}=k)=1}</annotation>
</semantics>
</math></span><img src="./e43de307488bca7f86f15d1e2b67c3d6ef065582.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -3.005ex; width:18.767ex; height:7.343ex;" alt="{\displaystyle \sum _{k=1}^{K}\Pr(Y_{i}=k)=1}" loading="lazy"></span></dd></dl>
<p>The reason why we need to add a term to ensure normalization, rather than multiply as is usual, is because we have taken the logarithm of the probabilities. Exponentiating both sides turns the additive term into a multiplicative factor, so that the probability is just the <a href="Gibbs_measure" title="Gibbs measure">Gibbs measure</a>:
</p>
<dl><dd><span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle \Pr(Y_{i}=k)={\frac {1}{Z}}e^{{\boldsymbol {\beta }}_{k}\cdot \mathbf {X} _{i}},\;\;\;\;\;\;1\leq k\leq K.}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mo movablelimits="true" form="prefix">Pr</mo>
<mo stretchy="false">(</mo>
<msub>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
<mo>=</mo>
<mi>k</mi>
<mo stretchy="false">)</mo>
<mo>=</mo>
<mrow class="MJX-TeXAtom-ORD">
<mfrac>
<mn>1</mn>
<mi>Z</mi>
</mfrac>
</mrow>
<msup>
<mi>e</mi>
<mrow class="MJX-TeXAtom-ORD">
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold-italic">β<!-- β --></mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>k</mi>
</mrow>
</msub>
<mo>⋅<!-- ⋅ --></mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold">X</mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
</mrow>
</msup>
<mo>,</mo>
<mspace width="thickmathspace"></mspace>
<mspace width="thickmathspace"></mspace>
<mspace width="thickmathspace"></mspace>
<mspace width="thickmathspace"></mspace>
<mspace width="thickmathspace"></mspace>
<mspace width="thickmathspace"></mspace>
<mn>1</mn>
<mo>≤<!-- ≤ --></mo>
<mi>k</mi>
<mo>≤<!-- ≤ --></mo>
<mi>K</mi>
<mo>.</mo>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle \Pr(Y_{i}=k)={\frac {1}{Z}}e^{{\boldsymbol {\beta }}_{k}\cdot \mathbf {X} _{i}},\;\;\;\;\;\;1\leq k\leq K.}</annotation>
</semantics>
</math></span><img src="./d499caf2089b06ea60726adaa3d89430cdad0059.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -1.838ex; width:38.337ex; height:5.176ex;" alt="{\displaystyle \Pr(Y_{i}=k)={\frac {1}{Z}}e^{{\boldsymbol {\beta }}_{k}\cdot \mathbf {X} _{i}},\;\;\;\;\;\;1\leq k\leq K.}" loading="lazy"></span></dd></dl>
<p>The quantity <i>Z</i> is called the <a href="Partition_function_(mathematics)" title="Partition function (mathematics)">partition function</a> for the distribution. We can compute the value of the partition function by applying the above constraint that requires all probabilities to sum to 1:
</p>
<dl><dd><span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle 1=\sum _{k=1}^{K}\Pr(Y_{i}=k)\;=\;\sum _{k=1}^{K}{\frac {1}{Z}}e^{{\boldsymbol {\beta }}_{k}\cdot \mathbf {X} _{i}}\;=\;{\frac {1}{Z}}\sum _{k=1}^{K}e^{{\boldsymbol {\beta }}_{k}\cdot \mathbf {X} _{i}}.}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mn>1</mn>
<mo>=</mo>
<munderover>
<mo>∑<!-- ∑ --></mo>
<mrow class="MJX-TeXAtom-ORD">
<mi>k</mi>
<mo>=</mo>
<mn>1</mn>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>K</mi>
</mrow>
</munderover>
<mo movablelimits="true" form="prefix">Pr</mo>
<mo stretchy="false">(</mo>
<msub>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
<mo>=</mo>
<mi>k</mi>
<mo stretchy="false">)</mo>
<mspace width="thickmathspace"></mspace>
<mo>=</mo>
<mspace width="thickmathspace"></mspace>
<munderover>
<mo>∑<!-- ∑ --></mo>
<mrow class="MJX-TeXAtom-ORD">
<mi>k</mi>
<mo>=</mo>
<mn>1</mn>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>K</mi>
</mrow>
</munderover>
<mrow class="MJX-TeXAtom-ORD">
<mfrac>
<mn>1</mn>
<mi>Z</mi>
</mfrac>
</mrow>
<msup>
<mi>e</mi>
<mrow class="MJX-TeXAtom-ORD">
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold-italic">β<!-- β --></mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>k</mi>
</mrow>
</msub>
<mo>⋅<!-- ⋅ --></mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold">X</mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
</mrow>
</msup>
<mspace width="thickmathspace"></mspace>
<mo>=</mo>
<mspace width="thickmathspace"></mspace>
<mrow class="MJX-TeXAtom-ORD">
<mfrac>
<mn>1</mn>
<mi>Z</mi>
</mfrac>
</mrow>
<munderover>
<mo>∑<!-- ∑ --></mo>
<mrow class="MJX-TeXAtom-ORD">
<mi>k</mi>
<mo>=</mo>
<mn>1</mn>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>K</mi>
</mrow>
</munderover>
<msup>
<mi>e</mi>
<mrow class="MJX-TeXAtom-ORD">
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold-italic">β<!-- β --></mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>k</mi>
</mrow>
</msub>
<mo>⋅<!-- ⋅ --></mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold">X</mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
</mrow>
</msup>
<mo>.</mo>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle 1=\sum _{k=1}^{K}\Pr(Y_{i}=k)\;=\;\sum _{k=1}^{K}{\frac {1}{Z}}e^{{\boldsymbol {\beta }}_{k}\cdot \mathbf {X} _{i}}\;=\;{\frac {1}{Z}}\sum _{k=1}^{K}e^{{\boldsymbol {\beta }}_{k}\cdot \mathbf {X} _{i}}.}</annotation>
</semantics>
</math></span><img src="./45d979a9a1dc9649763d6564e2d0cc673cd7cad2.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -3.005ex; width:52.636ex; height:7.343ex;" alt="{\displaystyle 1=\sum _{k=1}^{K}\Pr(Y_{i}=k)\;=\;\sum _{k=1}^{K}{\frac {1}{Z}}e^{{\boldsymbol {\beta }}_{k}\cdot \mathbf {X} _{i}}\;=\;{\frac {1}{Z}}\sum _{k=1}^{K}e^{{\boldsymbol {\beta }}_{k}\cdot \mathbf {X} _{i}}.}" loading="lazy"></span></dd></dl>
<p>Therefore
</p>
<dl><dd><span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle Z=\sum _{k=1}^{K}e^{{\boldsymbol {\beta }}_{k}\cdot \mathbf {X} _{i}}.}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>Z</mi>
<mo>=</mo>
<munderover>
<mo>∑<!-- ∑ --></mo>
<mrow class="MJX-TeXAtom-ORD">
<mi>k</mi>
<mo>=</mo>
<mn>1</mn>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>K</mi>
</mrow>
</munderover>
<msup>
<mi>e</mi>
<mrow class="MJX-TeXAtom-ORD">
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold-italic">β<!-- β --></mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>k</mi>
</mrow>
</msub>
<mo>⋅<!-- ⋅ --></mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold">X</mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
</mrow>
</msup>
<mo>.</mo>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle Z=\sum _{k=1}^{K}e^{{\boldsymbol {\beta }}_{k}\cdot \mathbf {X} _{i}}.}</annotation>
</semantics>
</math></span><img src="./065fd74572e55b9ef143ddea2d984abaf1c6d571.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -3.005ex; width:14.938ex; height:7.343ex;" alt="{\displaystyle Z=\sum _{k=1}^{K}e^{{\boldsymbol {\beta }}_{k}\cdot \mathbf {X} _{i}}.}" loading="lazy"></span></dd></dl>
<p>Note that this factor is "constant" in the sense that it is not a function of <i>Y</i><sub><i>i</i></sub>, which is the variable over which the probability distribution is defined. However, it is definitely not constant with respect to the explanatory variables, or crucially, with respect to the unknown regression coefficients <i><b>β</b></i><sub><i>k</i></sub>, which we will need to determine through some sort of <a href="Mathematical_optimization" title="Mathematical optimization">optimization</a> procedure.
</p><p>The resulting equations for the probabilities are
</p>
<dl><dd><span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle \Pr(Y_{i}=k)={\frac {e^{{\boldsymbol {\beta }}_{k}\cdot \mathbf {X} _{i}}}{\sum _{j=1}^{K}e^{{\boldsymbol {\beta }}_{j}\cdot \mathbf {X} _{i}}}},\;\;\;\;\;\;1\leq k\leq K.}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mo movablelimits="true" form="prefix">Pr</mo>
<mo stretchy="false">(</mo>
<msub>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
<mo>=</mo>
<mi>k</mi>
<mo stretchy="false">)</mo>
<mo>=</mo>
<mrow class="MJX-TeXAtom-ORD">
<mfrac>
<msup>
<mi>e</mi>
<mrow class="MJX-TeXAtom-ORD">
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold-italic">β<!-- β --></mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>k</mi>
</mrow>
</msub>
<mo>⋅<!-- ⋅ --></mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold">X</mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
</mrow>
</msup>
<mrow>
<munderover>
<mo>∑<!-- ∑ --></mo>
<mrow class="MJX-TeXAtom-ORD">
<mi>j</mi>
<mo>=</mo>
<mn>1</mn>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>K</mi>
</mrow>
</munderover>
<msup>
<mi>e</mi>
<mrow class="MJX-TeXAtom-ORD">
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold-italic">β<!-- β --></mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>j</mi>
</mrow>
</msub>
<mo>⋅<!-- ⋅ --></mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold">X</mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
</mrow>
</msup>
</mrow>
</mfrac>
</mrow>
<mo>,</mo>
<mspace width="thickmathspace"></mspace>
<mspace width="thickmathspace"></mspace>
<mspace width="thickmathspace"></mspace>
<mspace width="thickmathspace"></mspace>
<mspace width="thickmathspace"></mspace>
<mspace width="thickmathspace"></mspace>
<mn>1</mn>
<mo>≤<!-- ≤ --></mo>
<mi>k</mi>
<mo>≤<!-- ≤ --></mo>
<mi>K</mi>
<mo>.</mo>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle \Pr(Y_{i}=k)={\frac {e^{{\boldsymbol {\beta }}_{k}\cdot \mathbf {X} _{i}}}{\sum _{j=1}^{K}e^{{\boldsymbol {\beta }}_{j}\cdot \mathbf {X} _{i}}}},\;\;\;\;\;\;1\leq k\leq K.}</annotation>
</semantics>
</math></span><img src="./e8eb61deab2e13580d98c6020e7fdb33e9dea2bd.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -3.505ex; width:42.363ex; height:7.343ex;" alt="{\displaystyle \Pr(Y_{i}=k)={\frac {e^{{\boldsymbol {\beta }}_{k}\cdot \mathbf {X} _{i}}}{\sum _{j=1}^{K}e^{{\boldsymbol {\beta }}_{j}\cdot \mathbf {X} _{i}}}},\;\;\;\;\;\;1\leq k\leq K.}" loading="lazy"></span></dd></dl>
<p><br>
</p><p>The following function:
</p>
<dl><dd><span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle \operatorname {softmax} (k,s_{1},\ldots ,s_{K})={\frac {e^{s_{k}}}{\sum _{j=1}^{K}e^{s_{j}}}}}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>softmax</mi>
<mo>⁡<!-- ⁡ --></mo>
<mo stretchy="false">(</mo>
<mi>k</mi>
<mo>,</mo>
<msub>
<mi>s</mi>
<mrow class="MJX-TeXAtom-ORD">
<mn>1</mn>
</mrow>
</msub>
<mo>,</mo>
<mo>…<!-- … --></mo>
<mo>,</mo>
<msub>
<mi>s</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>K</mi>
</mrow>
</msub>
<mo stretchy="false">)</mo>
<mo>=</mo>
<mrow class="MJX-TeXAtom-ORD">
<mfrac>
<msup>
<mi>e</mi>
<mrow class="MJX-TeXAtom-ORD">
<msub>
<mi>s</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>k</mi>
</mrow>
</msub>
</mrow>
</msup>
<mrow>
<munderover>
<mo>∑<!-- ∑ --></mo>
<mrow class="MJX-TeXAtom-ORD">
<mi>j</mi>
<mo>=</mo>
<mn>1</mn>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>K</mi>
</mrow>
</munderover>
<msup>
<mi>e</mi>
<mrow class="MJX-TeXAtom-ORD">
<msub>
<mi>s</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>j</mi>
</mrow>
</msub>
</mrow>
</msup>
</mrow>
</mfrac>
</mrow>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle \operatorname {softmax} (k,s_{1},\ldots ,s_{K})={\frac {e^{s_{k}}}{\sum _{j=1}^{K}e^{s_{j}}}}}</annotation>
</semantics>
</math></span><img src="./92d2b05d1c4a1ee74e64b4f7542145170445d0d2.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -3.505ex; width:34.769ex; height:6.843ex;" alt="{\displaystyle \operatorname {softmax} (k,s_{1},\ldots ,s_{K})={\frac {e^{s_{k}}}{\sum _{j=1}^{K}e^{s_{j}}}}}" loading="lazy"></span></dd></dl>
<p>is referred to as the <a href="Softmax_function" title="Softmax function">softmax function</a>. The reason is that the effect of exponentiating the values <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle s_{1},\ldots ,s_{K}}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<msub>
<mi>s</mi>
<mrow class="MJX-TeXAtom-ORD">
<mn>1</mn>
</mrow>
</msub>
<mo>,</mo>
<mo>…<!-- … --></mo>
<mo>,</mo>
<msub>
<mi>s</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>K</mi>
</mrow>
</msub>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle s_{1},\ldots ,s_{K}}</annotation>
</semantics>
</math></span><img src="./1c865450f02fd46f380fef15f69c183d34626a19.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.671ex; width:10.106ex; height:2.009ex;" alt="{\displaystyle s_{1},\ldots ,s_{K}}" loading="lazy"></span> is to exaggerate the differences between them. As a result, <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle \operatorname {softmax} (k,s_{1},\ldots ,s_{K})}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>softmax</mi>
<mo>⁡<!-- ⁡ --></mo>
<mo stretchy="false">(</mo>
<mi>k</mi>
<mo>,</mo>
<msub>
<mi>s</mi>
<mrow class="MJX-TeXAtom-ORD">
<mn>1</mn>
</mrow>
</msub>
<mo>,</mo>
<mo>…<!-- … --></mo>
<mo>,</mo>
<msub>
<mi>s</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>K</mi>
</mrow>
</msub>
<mo stretchy="false">)</mo>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle \operatorname {softmax} (k,s_{1},\ldots ,s_{K})}</annotation>
</semantics>
</math></span><img src="./534e56fb34000b2aebb71c72a5f06cb18e86dbe4.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.838ex; width:22.182ex; height:2.843ex;" alt="{\displaystyle \operatorname {softmax} (k,s_{1},\ldots ,s_{K})}" loading="lazy"></span> will return a value close to 0 whenever <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle s_{k}}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<msub>
<mi>s</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>k</mi>
</mrow>
</msub>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle s_{k}}</annotation>
</semantics>
</math></span><img src="./04f159343172781e7666dbc88280c91f34117c30.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.671ex; width:2.179ex; height:2.009ex;" alt="{\displaystyle s_{k}}" loading="lazy"></span> is significantly less than the maximum of all the values, and will return a value close to 1 when applied to the maximum value, unless it is extremely close to the next-largest value. Thus, the softmax function can be used to construct a <a href="Weighted_average" title="Weighted average">weighted average</a> that behaves as a <a href="Smooth_function" class="mw-redirect" title="Smooth function">smooth function</a> (which can be conveniently <a href="Differentiation_(mathematics)" class="mw-redirect" title="Differentiation (mathematics)">differentiated</a>, etc.) and which approximates the <a href="Indicator_function" title="Indicator function">indicator function</a>
</p>
<dl><dd><span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle f(k)={\begin{cases}1&amp;{\textrm {if}}\;k=\operatorname {\arg \max } _{j}s_{j},\\0&amp;{\textrm {otherwise}}.\end{cases}}}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>f</mi>
<mo stretchy="false">(</mo>
<mi>k</mi>
<mo stretchy="false">)</mo>
<mo>=</mo>
<mrow class="MJX-TeXAtom-ORD">
<mrow>
<mo>{</mo>
<mtable columnalign="left left" rowspacing=".2em" columnspacing="1em" displaystyle="false">
<mtr>
<mtd>
<mn>1</mn>
</mtd>
<mtd>
<mrow class="MJX-TeXAtom-ORD">
<mrow class="MJX-TeXAtom-ORD">
<mtext>if</mtext>
</mrow>
</mrow>
<mspace width="thickmathspace"></mspace>
<mi>k</mi>
<mo>=</mo>
<msub>
<mrow class="MJX-TeXAtom-OP MJX-fixedlimits">
<mi>arg</mi>
<mo>⁡<!-- ⁡ --></mo>
<mo movablelimits="true" form="prefix">max</mo>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>j</mi>
</mrow>
</msub>
<mo>⁡<!-- ⁡ --></mo>
<msub>
<mi>s</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>j</mi>
</mrow>
</msub>
<mo>,</mo>
</mtd>
</mtr>
<mtr>
<mtd>
<mn>0</mn>
</mtd>
<mtd>
<mrow class="MJX-TeXAtom-ORD">
<mrow class="MJX-TeXAtom-ORD">
<mtext>otherwise</mtext>
</mrow>
</mrow>
<mo>.</mo>
</mtd>
</mtr>
</mtable>
<mo fence="true" stretchy="true" symmetric="true"></mo>
</mrow>
</mrow>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle f(k)={\begin{cases}1&amp;{\textrm {if}}\;k=\operatorname {\arg \max } _{j}s_{j},\\0&amp;{\textrm {otherwise}}.\end{cases}}}</annotation>
</semantics>
</math></span><img src="./abdec8770a48f9d2b9167ff37a9045caf786449c.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -2.505ex; width:31.584ex; height:6.176ex;" alt="{\displaystyle f(k)={\begin{cases}1&amp;{\textrm {if}}\;k=\operatorname {\arg \max } _{j}s_{j},\\0&amp;{\textrm {otherwise}}.\end{cases}}}" loading="lazy"></span></dd></dl>
<p>Thus, we can write the probability equations as
</p>
<dl><dd><span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle \Pr(Y_{i}=k)=\operatorname {softmax} (k,{\boldsymbol {\beta }}_{1}\cdot \mathbf {X} _{i},\ldots ,{\boldsymbol {\beta }}_{K}\cdot \mathbf {X} _{i})}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mo movablelimits="true" form="prefix">Pr</mo>
<mo stretchy="false">(</mo>
<msub>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
<mo>=</mo>
<mi>k</mi>
<mo stretchy="false">)</mo>
<mo>=</mo>
<mi>softmax</mi>
<mo>⁡<!-- ⁡ --></mo>
<mo stretchy="false">(</mo>
<mi>k</mi>
<mo>,</mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold-italic">β<!-- β --></mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mn>1</mn>
</mrow>
</msub>
<mo>⋅<!-- ⋅ --></mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold">X</mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
<mo>,</mo>
<mo>…<!-- … --></mo>
<mo>,</mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold-italic">β<!-- β --></mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>K</mi>
</mrow>
</msub>
<mo>⋅<!-- ⋅ --></mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold">X</mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
<mo stretchy="false">)</mo>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle \Pr(Y_{i}=k)=\operatorname {softmax} (k,{\boldsymbol {\beta }}_{1}\cdot \mathbf {X} _{i},\ldots ,{\boldsymbol {\beta }}_{K}\cdot \mathbf {X} _{i})}</annotation>
</semantics>
</math></span><img src="./0cae9a7e99bd0fc580fc7bddd22d1d4582c71715.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.838ex; width:45.928ex; height:2.843ex;" alt="{\displaystyle \Pr(Y_{i}=k)=\operatorname {softmax} (k,{\boldsymbol {\beta }}_{1}\cdot \mathbf {X} _{i},\ldots ,{\boldsymbol {\beta }}_{K}\cdot \mathbf {X} _{i})}" loading="lazy"></span></dd></dl>
<p>The softmax function thus serves as the equivalent of the <a href="Logistic_function" title="Logistic function">logistic function</a> in binary logistic regression.
</p><p>Note that not all of the <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle {\boldsymbol {\beta }}_{k}}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold-italic">β<!-- β --></mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>k</mi>
</mrow>
</msub>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle {\boldsymbol {\beta }}_{k}}</annotation>
</semantics>
</math></span><img src="./07f1bcb6340535f54accaf88124f4cd5ba1d4fc8.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.838ex; width:2.623ex; height:2.676ex;" alt="{\displaystyle {\boldsymbol {\beta }}_{k}}" loading="lazy"></span> vectors of coefficients are uniquely <a href="Identifiability" title="Identifiability">identifiable</a>. This is due to the fact that all probabilities must sum to 1, making one of them completely determined once all the rest are known. As a result, there are only <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle K-1}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>K</mi>
<mo>−<!-- − --></mo>
<mn>1</mn>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle K-1}</annotation>
</semantics>
</math></span><img src="./dd6e429474c269979f75b41db5d334243b3dccd3.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.505ex; width:6.069ex; height:2.343ex;" alt="{\displaystyle K-1}" loading="lazy"></span> separately specifiable probabilities, and hence <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle K-1}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>K</mi>
<mo>−<!-- − --></mo>
<mn>1</mn>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle K-1}</annotation>
</semantics>
</math></span><img src="./dd6e429474c269979f75b41db5d334243b3dccd3.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.505ex; width:6.069ex; height:2.343ex;" alt="{\displaystyle K-1}" loading="lazy"></span> separately identifiable vectors of coefficients. One way to see this is to note that if we add a constant vector to all of the coefficient vectors, the equations are identical:
</p>
<dl><dd><span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle {\begin{aligned}{\frac {e^{({\boldsymbol {\beta }}_{k}+\mathbf {C} )\cdot \mathbf {X} _{i}}}{\sum _{j=1}^{K}e^{({\boldsymbol {\beta }}_{j}+\mathbf {C} )\cdot \mathbf {X} _{i}}}}&amp;={\frac {e^{{\boldsymbol {\beta }}_{k}\cdot \mathbf {X} _{i}}e^{\mathbf {C} \cdot \mathbf {X} _{i}}}{\sum _{j=1}^{K}e^{{\boldsymbol {\beta }}_{j}\cdot \mathbf {X} _{i}}e^{\mathbf {C} \cdot \mathbf {X} _{i}}}}\\&amp;={\frac {e^{\mathbf {C} \cdot \mathbf {X} _{i}}e^{{\boldsymbol {\beta }}_{k}\cdot \mathbf {X} _{i}}}{e^{\mathbf {C} \cdot \mathbf {X} _{i}}\sum _{j=1}^{K}e^{{\boldsymbol {\beta }}_{j}\cdot \mathbf {X} _{i}}}}\\&amp;={\frac {e^{{\boldsymbol {\beta }}_{k}\cdot \mathbf {X} _{i}}}{\sum _{j=1}^{K}e^{{\boldsymbol {\beta }}_{j}\cdot \mathbf {X} _{i}}}}\end{aligned}}}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mrow class="MJX-TeXAtom-ORD">
<mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true">
<mtr>
<mtd>
<mrow class="MJX-TeXAtom-ORD">
<mfrac>
<msup>
<mi>e</mi>
<mrow class="MJX-TeXAtom-ORD">
<mo stretchy="false">(</mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold-italic">β<!-- β --></mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>k</mi>
</mrow>
</msub>
<mo>+</mo>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold">C</mi>
</mrow>
<mo stretchy="false">)</mo>
<mo>⋅<!-- ⋅ --></mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold">X</mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
</mrow>
</msup>
<mrow>
<munderover>
<mo>∑<!-- ∑ --></mo>
<mrow class="MJX-TeXAtom-ORD">
<mi>j</mi>
<mo>=</mo>
<mn>1</mn>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>K</mi>
</mrow>
</munderover>
<msup>
<mi>e</mi>
<mrow class="MJX-TeXAtom-ORD">
<mo stretchy="false">(</mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold-italic">β<!-- β --></mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>j</mi>
</mrow>
</msub>
<mo>+</mo>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold">C</mi>
</mrow>
<mo stretchy="false">)</mo>
<mo>⋅<!-- ⋅ --></mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold">X</mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
</mrow>
</msup>
</mrow>
</mfrac>
</mrow>
</mtd>
<mtd>
<mi></mi>
<mo>=</mo>
<mrow class="MJX-TeXAtom-ORD">
<mfrac>
<mrow>
<msup>
<mi>e</mi>
<mrow class="MJX-TeXAtom-ORD">
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold-italic">β<!-- β --></mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>k</mi>
</mrow>
</msub>
<mo>⋅<!-- ⋅ --></mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold">X</mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
</mrow>
</msup>
<msup>
<mi>e</mi>
<mrow class="MJX-TeXAtom-ORD">
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold">C</mi>
</mrow>
<mo>⋅<!-- ⋅ --></mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold">X</mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
</mrow>
</msup>
</mrow>
<mrow>
<munderover>
<mo>∑<!-- ∑ --></mo>
<mrow class="MJX-TeXAtom-ORD">
<mi>j</mi>
<mo>=</mo>
<mn>1</mn>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>K</mi>
</mrow>
</munderover>
<msup>
<mi>e</mi>
<mrow class="MJX-TeXAtom-ORD">
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold-italic">β<!-- β --></mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>j</mi>
</mrow>
</msub>
<mo>⋅<!-- ⋅ --></mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold">X</mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
</mrow>
</msup>
<msup>
<mi>e</mi>
<mrow class="MJX-TeXAtom-ORD">
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold">C</mi>
</mrow>
<mo>⋅<!-- ⋅ --></mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold">X</mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
</mrow>
</msup>
</mrow>
</mfrac>
</mrow>
</mtd>
</mtr>
<mtr>
<mtd></mtd>
<mtd>
<mi></mi>
<mo>=</mo>
<mrow class="MJX-TeXAtom-ORD">
<mfrac>
<mrow>
<msup>
<mi>e</mi>
<mrow class="MJX-TeXAtom-ORD">
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold">C</mi>
</mrow>
<mo>⋅<!-- ⋅ --></mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold">X</mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
</mrow>
</msup>
<msup>
<mi>e</mi>
<mrow class="MJX-TeXAtom-ORD">
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold-italic">β<!-- β --></mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>k</mi>
</mrow>
</msub>
<mo>⋅<!-- ⋅ --></mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold">X</mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
</mrow>
</msup>
</mrow>
<mrow>
<msup>
<mi>e</mi>
<mrow class="MJX-TeXAtom-ORD">
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold">C</mi>
</mrow>
<mo>⋅<!-- ⋅ --></mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold">X</mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
</mrow>
</msup>
<munderover>
<mo>∑<!-- ∑ --></mo>
<mrow class="MJX-TeXAtom-ORD">
<mi>j</mi>
<mo>=</mo>
<mn>1</mn>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>K</mi>
</mrow>
</munderover>
<msup>
<mi>e</mi>
<mrow class="MJX-TeXAtom-ORD">
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold-italic">β<!-- β --></mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>j</mi>
</mrow>
</msub>
<mo>⋅<!-- ⋅ --></mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold">X</mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
</mrow>
</msup>
</mrow>
</mfrac>
</mrow>
</mtd>
</mtr>
<mtr>
<mtd></mtd>
<mtd>
<mi></mi>
<mo>=</mo>
<mrow class="MJX-TeXAtom-ORD">
<mfrac>
<msup>
<mi>e</mi>
<mrow class="MJX-TeXAtom-ORD">
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold-italic">β<!-- β --></mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>k</mi>
</mrow>
</msub>
<mo>⋅<!-- ⋅ --></mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold">X</mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
</mrow>
</msup>
<mrow>
<munderover>
<mo>∑<!-- ∑ --></mo>
<mrow class="MJX-TeXAtom-ORD">
<mi>j</mi>
<mo>=</mo>
<mn>1</mn>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>K</mi>
</mrow>
</munderover>
<msup>
<mi>e</mi>
<mrow class="MJX-TeXAtom-ORD">
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold-italic">β<!-- β --></mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>j</mi>
</mrow>
</msub>
<mo>⋅<!-- ⋅ --></mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold">X</mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
</mrow>
</msup>
</mrow>
</mfrac>
</mrow>
</mtd>
</mtr>
</mtable>
</mrow>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle {\begin{aligned}{\frac {e^{({\boldsymbol {\beta }}_{k}+\mathbf {C} )\cdot \mathbf {X} _{i}}}{\sum _{j=1}^{K}e^{({\boldsymbol {\beta }}_{j}+\mathbf {C} )\cdot \mathbf {X} _{i}}}}&amp;={\frac {e^{{\boldsymbol {\beta }}_{k}\cdot \mathbf {X} _{i}}e^{\mathbf {C} \cdot \mathbf {X} _{i}}}{\sum _{j=1}^{K}e^{{\boldsymbol {\beta }}_{j}\cdot \mathbf {X} _{i}}e^{\mathbf {C} \cdot \mathbf {X} _{i}}}}\\&amp;={\frac {e^{\mathbf {C} \cdot \mathbf {X} _{i}}e^{{\boldsymbol {\beta }}_{k}\cdot \mathbf {X} _{i}}}{e^{\mathbf {C} \cdot \mathbf {X} _{i}}\sum _{j=1}^{K}e^{{\boldsymbol {\beta }}_{j}\cdot \mathbf {X} _{i}}}}\\&amp;={\frac {e^{{\boldsymbol {\beta }}_{k}\cdot \mathbf {X} _{i}}}{\sum _{j=1}^{K}e^{{\boldsymbol {\beta }}_{j}\cdot \mathbf {X} _{i}}}}\end{aligned}}}</annotation>
</semantics>
</math></span><img src="./1abb5acc4ac5305a682d78ae51b3c150c289eca7.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -10.671ex; width:37.977ex; height:22.509ex;" alt="{\displaystyle {\begin{aligned}{\frac {e^{({\boldsymbol {\beta }}_{k}+\mathbf {C} )\cdot \mathbf {X} _{i}}}{\sum _{j=1}^{K}e^{({\boldsymbol {\beta }}_{j}+\mathbf {C} )\cdot \mathbf {X} _{i}}}}&amp;={\frac {e^{{\boldsymbol {\beta }}_{k}\cdot \mathbf {X} _{i}}e^{\mathbf {C} \cdot \mathbf {X} _{i}}}{\sum _{j=1}^{K}e^{{\boldsymbol {\beta }}_{j}\cdot \mathbf {X} _{i}}e^{\mathbf {C} \cdot \mathbf {X} _{i}}}}\\&amp;={\frac {e^{\mathbf {C} \cdot \mathbf {X} _{i}}e^{{\boldsymbol {\beta }}_{k}\cdot \mathbf {X} _{i}}}{e^{\mathbf {C} \cdot \mathbf {X} _{i}}\sum _{j=1}^{K}e^{{\boldsymbol {\beta }}_{j}\cdot \mathbf {X} _{i}}}}\\&amp;={\frac {e^{{\boldsymbol {\beta }}_{k}\cdot \mathbf {X} _{i}}}{\sum _{j=1}^{K}e^{{\boldsymbol {\beta }}_{j}\cdot \mathbf {X} _{i}}}}\end{aligned}}}" loading="lazy"></span></dd></dl>
<p>As a result, it is conventional to set <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle \mathbf {C} =-{\boldsymbol {\beta }}_{K}}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold">C</mi>
</mrow>
<mo>=</mo>
<mo>−<!-- − --></mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold-italic">β<!-- β --></mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>K</mi>
</mrow>
</msub>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle \mathbf {C} =-{\boldsymbol {\beta }}_{K}}</annotation>
</semantics>
</math></span><img src="./345cf4cbde600926afdbc72161a6c8d05cd9a89e.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.838ex; width:10.065ex; height:2.676ex;" alt="{\displaystyle \mathbf {C} =-{\boldsymbol {\beta }}_{K}}" loading="lazy"></span> (or alternatively, one of the other coefficient vectors). Essentially, we set the constant so that one of the vectors becomes <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle {\boldsymbol {0}}}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mrow class="MJX-TeXAtom-ORD">
<mn mathvariant="bold">0</mn>
</mrow>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle {\boldsymbol {0}}}</annotation>
</semantics>
</math></span><img src="./58fe04d1d35aac19861a0d7c9f7c374c88d42db0.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.338ex; width:1.337ex; height:2.176ex;" alt="{\displaystyle {\boldsymbol {0}}}" loading="lazy"></span>, and all of the other vectors get transformed into the difference between those vectors and the vector we chose. This is equivalent to "pivoting" around one of the <i>K</i> choices, and examining how much better or worse all of the other <i>K</i>&nbsp;−&nbsp;1 choices are, relative to the choice we are pivoting around. Mathematically, we transform the coefficients as follows:
</p>
<dl><dd><span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle {\begin{aligned}{\boldsymbol {\beta }}'_{k}&amp;={\boldsymbol {\beta }}_{k}-{\boldsymbol {\beta }}_{K},\;\;\;\;1\leq k<K,\\{\boldsymbol {\beta }}'_{K}&amp;=0.\end{aligned}}}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mrow class="MJX-TeXAtom-ORD">
<mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true">
<mtr>
<mtd>
<msubsup>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold-italic">β<!-- β --></mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>k</mi>
</mrow>
<mo>′</mo>
</msubsup>
</mtd>
<mtd>
<mi></mi>
<mo>=</mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold-italic">β<!-- β --></mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>k</mi>
</mrow>
</msub>
<mo>−<!-- − --></mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold-italic">β<!-- β --></mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>K</mi>
</mrow>
</msub>
<mo>,</mo>
<mspace width="thickmathspace"></mspace>
<mspace width="thickmathspace"></mspace>
<mspace width="thickmathspace"></mspace>
<mspace width="thickmathspace"></mspace>
<mn>1</mn>
<mo>≤<!-- ≤ --></mo>
<mi>k</mi>
<mo>&lt;</mo>
<mi>K</mi>
<mo>,</mo>
</mtd>
</mtr>
<mtr>
<mtd>
<msubsup>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold-italic">β<!-- β --></mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>K</mi>
</mrow>
<mo>′</mo>
</msubsup>
</mtd>
<mtd>
<mi></mi>
<mo>=</mo>
<mn>0.</mn>
</mtd>
</mtr>
</mtable>
</mrow>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle {\begin{aligned}{\boldsymbol {\beta }}'_{k}&amp;={\boldsymbol {\beta }}_{k}-{\boldsymbol {\beta }}_{K},\;\;\;\;1\leq k&lt;K,\\{\boldsymbol {\beta }}'_{K}&amp;=0.\end{aligned}}}</annotation>
</semantics>
</math></span><img src="./db8474b3fba9afa0f82227175b8b19c0e038bdcc.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -2.505ex; width:30.665ex; height:6.176ex;" alt="{\displaystyle {\begin{aligned}{\boldsymbol {\beta }}'_{k}&amp;={\boldsymbol {\beta }}_{k}-{\boldsymbol {\beta }}_{K},\;\;\;\;1\leq k<K,\\{\boldsymbol {\beta }}'_{K}&amp;=0.\end{aligned}}}" loading="lazy"></span></dd></dl>
<p>This leads to the following equations:
</p>
<dl><dd><span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle \Pr(Y_{i}=k)={\frac {e^{{\boldsymbol {\beta }}'_{k}\cdot \mathbf {X} _{i}}}{1+\sum _{j=1}^{K-1}e^{{\boldsymbol {\beta }}'_{j}\cdot \mathbf {X} _{i}}}},\;\;\;\;\;\;1\leq k\leq K}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mo movablelimits="true" form="prefix">Pr</mo>
<mo stretchy="false">(</mo>
<msub>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
<mo>=</mo>
<mi>k</mi>
<mo stretchy="false">)</mo>
<mo>=</mo>
<mrow class="MJX-TeXAtom-ORD">
<mfrac>
<msup>
<mi>e</mi>
<mrow class="MJX-TeXAtom-ORD">
<msubsup>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold-italic">β<!-- β --></mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>k</mi>
</mrow>
<mo>′</mo>
</msubsup>
<mo>⋅<!-- ⋅ --></mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold">X</mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
</mrow>
</msup>
<mrow>
<mn>1</mn>
<mo>+</mo>
<munderover>
<mo>∑<!-- ∑ --></mo>
<mrow class="MJX-TeXAtom-ORD">
<mi>j</mi>
<mo>=</mo>
<mn>1</mn>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>K</mi>
<mo>−<!-- − --></mo>
<mn>1</mn>
</mrow>
</munderover>
<msup>
<mi>e</mi>
<mrow class="MJX-TeXAtom-ORD">
<msubsup>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold-italic">β<!-- β --></mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>j</mi>
</mrow>
<mo>′</mo>
</msubsup>
<mo>⋅<!-- ⋅ --></mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold">X</mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
</mrow>
</msup>
</mrow>
</mfrac>
</mrow>
<mo>,</mo>
<mspace width="thickmathspace"></mspace>
<mspace width="thickmathspace"></mspace>
<mspace width="thickmathspace"></mspace>
<mspace width="thickmathspace"></mspace>
<mspace width="thickmathspace"></mspace>
<mspace width="thickmathspace"></mspace>
<mn>1</mn>
<mo>≤<!-- ≤ --></mo>
<mi>k</mi>
<mo>≤<!-- ≤ --></mo>
<mi>K</mi>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle \Pr(Y_{i}=k)={\frac {e^{{\boldsymbol {\beta }}'_{k}\cdot \mathbf {X} _{i}}}{1+\sum _{j=1}^{K-1}e^{{\boldsymbol {\beta }}'_{j}\cdot \mathbf {X} _{i}}}},\;\;\;\;\;\;1\leq k\leq K}</annotation>
</semantics>
</math></span><img src="./fc7cf433154d2d75dd1b85d14e6ccb34fc6180aa.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -3.671ex; width:46.502ex; height:7.676ex;" alt="{\displaystyle \Pr(Y_{i}=k)={\frac {e^{{\boldsymbol {\beta }}'_{k}\cdot \mathbf {X} _{i}}}{1+\sum _{j=1}^{K-1}e^{{\boldsymbol {\beta }}'_{j}\cdot \mathbf {X} _{i}}}},\;\;\;\;\;\;1\leq k\leq K}" loading="lazy"></span></dd></dl>
<p>Other than the prime symbols on the regression coefficients, this is exactly the same as the form of the model described above, in terms of <i>K</i>&nbsp;−&nbsp;1 independent two-way regressions.
</p>
<div class="mw-heading mw-heading3"><h3 id="As_a_latent-variable_model">As a latent-variable model</h3></div>
<p>It is also possible to formulate multinomial logistic regression as a latent variable model, following the <a href="Logistic_regression#Two-way_latent-variable_model" title="Logistic regression">two-way latent variable model</a> described for binary logistic regression. This formulation is common in the theory of <a href="Discrete_choice" title="Discrete choice">discrete choice</a> models, and makes it easier to compare multinomial logistic regression to the related <a href="Multinomial_probit" title="Multinomial probit">multinomial probit</a> model, as well as to extend it to more complex models.
</p><p>Imagine that, for each data point <i>i</i> and possible outcome <i>k</i>&nbsp;=&nbsp;1,2,...,<i>K</i>, there is a continuous <a href="Latent_variable" class="mw-redirect" title="Latent variable">latent variable</a> <i>Y</i><sub><i>i,k</i></sub><sup><i>*</i></sup> (i.e. an unobserved <a href="Random_variable" title="Random variable">random variable</a>) that is distributed as follows:
</p>
<dl><dd><span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle Y_{i,k}^{\ast }={\boldsymbol {\beta }}_{k}\cdot \mathbf {X} _{i}+\varepsilon _{k}\;\;\;\;,\;\;k\leq K}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<msubsup>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
<mo>,</mo>
<mi>k</mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mo>∗<!-- ∗ --></mo>
</mrow>
</msubsup>
<mo>=</mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold-italic">β<!-- β --></mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>k</mi>
</mrow>
</msub>
<mo>⋅<!-- ⋅ --></mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold">X</mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
<mo>+</mo>
<msub>
<mi>ε<!-- ε --></mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>k</mi>
</mrow>
</msub>
<mspace width="thickmathspace"></mspace>
<mspace width="thickmathspace"></mspace>
<mspace width="thickmathspace"></mspace>
<mspace width="thickmathspace"></mspace>
<mo>,</mo>
<mspace width="thickmathspace"></mspace>
<mspace width="thickmathspace"></mspace>
<mi>k</mi>
<mo>≤<!-- ≤ --></mo>
<mi>K</mi>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle Y_{i,k}^{\ast }={\boldsymbol {\beta }}_{k}\cdot \mathbf {X} _{i}+\varepsilon _{k}\;\;\;\;,\;\;k\leq K}</annotation>
</semantics>
</math></span><img src="./1cf7613de4b63da8edb4eb0ff2b16cb5d391c148.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -1.338ex; width:29.977ex; height:3.176ex;" alt="{\displaystyle Y_{i,k}^{\ast }={\boldsymbol {\beta }}_{k}\cdot \mathbf {X} _{i}+\varepsilon _{k}\;\;\;\;,\;\;k\leq K}" loading="lazy"></span></dd></dl>
<p>where <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle \varepsilon _{k}\sim \operatorname {EV} _{1}(0,1),}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<msub>
<mi>ε<!-- ε --></mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>k</mi>
</mrow>
</msub>
<mo>∼<!-- ∼ --></mo>
<msub>
<mi>EV</mi>
<mrow class="MJX-TeXAtom-ORD">
<mn>1</mn>
</mrow>
</msub>
<mo>⁡<!-- ⁡ --></mo>
<mo stretchy="false">(</mo>
<mn>0</mn>
<mo>,</mo>
<mn>1</mn>
<mo stretchy="false">)</mo>
<mo>,</mo>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle \varepsilon _{k}\sim \operatorname {EV} _{1}(0,1),}</annotation>
</semantics>
</math></span><img src="./41de383261fa14bcd308d2390fbb28f6e3ce35e1.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.838ex; width:15.466ex; height:2.843ex;" alt="{\displaystyle \varepsilon _{k}\sim \operatorname {EV} _{1}(0,1),}" loading="lazy"></span> i.e. a standard type-1 <a href="Extreme_value_distribution" class="mw-redirect" title="Extreme value distribution">extreme value distribution</a>.
</p><p>This latent variable can be thought of as the <a href="Utility" title="Utility">utility</a> associated with data point <i>i</i> choosing outcome <i>k</i>, where there is some randomness in the actual amount of utility obtained, which accounts for other unmodeled factors that go into the choice. The value of the actual variable <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle Y_{i}}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<msub>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle Y_{i}}</annotation>
</semantics>
</math></span><img src="./d57be496fff95ee2a97ee43c7f7fe244b4dbf8ae.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.671ex; width:2.15ex; height:2.509ex;" alt="{\displaystyle Y_{i}}" loading="lazy"></span> is then determined in a non-random fashion from these latent variables (i.e. the randomness has been moved from the observed outcomes into the latent variables), where outcome <i>k</i> is chosen <a href="If_and_only_if" title="If and only if">if and only if</a> the associated utility (the value of <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle Y_{i,k}^{\ast }}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<msubsup>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
<mo>,</mo>
<mi>k</mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mo>∗<!-- ∗ --></mo>
</mrow>
</msubsup>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle Y_{i,k}^{\ast }}</annotation>
</semantics>
</math></span><img src="./33eb7f636a5049b93f92a3eed751e0533948f07a.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -1.338ex; width:3.464ex; height:3.176ex;" alt="{\displaystyle Y_{i,k}^{\ast }}" loading="lazy"></span>) is greater than the utilities of all the other choices, i.e. if the utility associated with outcome <i>k</i> is the maximum of all the utilities. Since the latent variables are <a href="Continuous_variable" class="mw-redirect" title="Continuous variable">continuous</a>, the probability of two having exactly the same value is 0, so we ignore the scenario. That is:
</p>
<dl><dd><span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle {\begin{aligned}\Pr(Y_{i}=1)&amp;=\Pr(Y_{i,1}^{\ast }>Y_{i,2}^{\ast }{\text{ and }}Y_{i,1}^{\ast }>Y_{i,3}^{\ast }{\text{ and }}\cdots {\text{ and }}Y_{i,1}^{\ast }>Y_{i,K}^{\ast })\\\Pr(Y_{i}=2)&amp;=\Pr(Y_{i,2}^{\ast }>Y_{i,1}^{\ast }{\text{ and }}Y_{i,2}^{\ast }>Y_{i,3}^{\ast }{\text{ and }}\cdots {\text{ and }}Y_{i,2}^{\ast }>Y_{i,K}^{\ast })\\&amp;\,\,\,\vdots \\\Pr(Y_{i}=K)&amp;=\Pr(Y_{i,K}^{\ast }>Y_{i,1}^{\ast }{\text{ and }}Y_{i,K}^{\ast }>Y_{i,2}^{\ast }{\text{ and }}\cdots {\text{ and }}Y_{i,K}^{\ast }>Y_{i,K-1}^{\ast })\\\end{aligned}}}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mrow class="MJX-TeXAtom-ORD">
<mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true">
<mtr>
<mtd>
<mo movablelimits="true" form="prefix">Pr</mo>
<mo stretchy="false">(</mo>
<msub>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
<mo>=</mo>
<mn>1</mn>
<mo stretchy="false">)</mo>
</mtd>
<mtd>
<mi></mi>
<mo>=</mo>
<mo movablelimits="true" form="prefix">Pr</mo>
<mo stretchy="false">(</mo>
<msubsup>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
<mo>,</mo>
<mn>1</mn>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mo>∗<!-- ∗ --></mo>
</mrow>
</msubsup>
<mo>&gt;</mo>
<msubsup>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
<mo>,</mo>
<mn>2</mn>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mo>∗<!-- ∗ --></mo>
</mrow>
</msubsup>
<mrow class="MJX-TeXAtom-ORD">
<mtext>&nbsp;and&nbsp;</mtext>
</mrow>
<msubsup>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
<mo>,</mo>
<mn>1</mn>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mo>∗<!-- ∗ --></mo>
</mrow>
</msubsup>
<mo>&gt;</mo>
<msubsup>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
<mo>,</mo>
<mn>3</mn>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mo>∗<!-- ∗ --></mo>
</mrow>
</msubsup>
<mrow class="MJX-TeXAtom-ORD">
<mtext>&nbsp;and&nbsp;</mtext>
</mrow>
<mo>⋯<!-- ⋯ --></mo>
<mrow class="MJX-TeXAtom-ORD">
<mtext>&nbsp;and&nbsp;</mtext>
</mrow>
<msubsup>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
<mo>,</mo>
<mn>1</mn>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mo>∗<!-- ∗ --></mo>
</mrow>
</msubsup>
<mo>&gt;</mo>
<msubsup>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
<mo>,</mo>
<mi>K</mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mo>∗<!-- ∗ --></mo>
</mrow>
</msubsup>
<mo stretchy="false">)</mo>
</mtd>
</mtr>
<mtr>
<mtd>
<mo movablelimits="true" form="prefix">Pr</mo>
<mo stretchy="false">(</mo>
<msub>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
<mo>=</mo>
<mn>2</mn>
<mo stretchy="false">)</mo>
</mtd>
<mtd>
<mi></mi>
<mo>=</mo>
<mo movablelimits="true" form="prefix">Pr</mo>
<mo stretchy="false">(</mo>
<msubsup>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
<mo>,</mo>
<mn>2</mn>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mo>∗<!-- ∗ --></mo>
</mrow>
</msubsup>
<mo>&gt;</mo>
<msubsup>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
<mo>,</mo>
<mn>1</mn>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mo>∗<!-- ∗ --></mo>
</mrow>
</msubsup>
<mrow class="MJX-TeXAtom-ORD">
<mtext>&nbsp;and&nbsp;</mtext>
</mrow>
<msubsup>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
<mo>,</mo>
<mn>2</mn>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mo>∗<!-- ∗ --></mo>
</mrow>
</msubsup>
<mo>&gt;</mo>
<msubsup>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
<mo>,</mo>
<mn>3</mn>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mo>∗<!-- ∗ --></mo>
</mrow>
</msubsup>
<mrow class="MJX-TeXAtom-ORD">
<mtext>&nbsp;and&nbsp;</mtext>
</mrow>
<mo>⋯<!-- ⋯ --></mo>
<mrow class="MJX-TeXAtom-ORD">
<mtext>&nbsp;and&nbsp;</mtext>
</mrow>
<msubsup>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
<mo>,</mo>
<mn>2</mn>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mo>∗<!-- ∗ --></mo>
</mrow>
</msubsup>
<mo>&gt;</mo>
<msubsup>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
<mo>,</mo>
<mi>K</mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mo>∗<!-- ∗ --></mo>
</mrow>
</msubsup>
<mo stretchy="false">)</mo>
</mtd>
</mtr>
<mtr>
<mtd></mtd>
<mtd>
<mi></mi>
<mspace width="thinmathspace"></mspace>
<mspace width="thinmathspace"></mspace>
<mspace width="thinmathspace"></mspace>
<mo>⋮<!-- ⋮ --></mo>
</mtd>
</mtr>
<mtr>
<mtd>
<mo movablelimits="true" form="prefix">Pr</mo>
<mo stretchy="false">(</mo>
<msub>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
<mo>=</mo>
<mi>K</mi>
<mo stretchy="false">)</mo>
</mtd>
<mtd>
<mi></mi>
<mo>=</mo>
<mo movablelimits="true" form="prefix">Pr</mo>
<mo stretchy="false">(</mo>
<msubsup>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
<mo>,</mo>
<mi>K</mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mo>∗<!-- ∗ --></mo>
</mrow>
</msubsup>
<mo>&gt;</mo>
<msubsup>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
<mo>,</mo>
<mn>1</mn>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mo>∗<!-- ∗ --></mo>
</mrow>
</msubsup>
<mrow class="MJX-TeXAtom-ORD">
<mtext>&nbsp;and&nbsp;</mtext>
</mrow>
<msubsup>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
<mo>,</mo>
<mi>K</mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mo>∗<!-- ∗ --></mo>
</mrow>
</msubsup>
<mo>&gt;</mo>
<msubsup>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
<mo>,</mo>
<mn>2</mn>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mo>∗<!-- ∗ --></mo>
</mrow>
</msubsup>
<mrow class="MJX-TeXAtom-ORD">
<mtext>&nbsp;and&nbsp;</mtext>
</mrow>
<mo>⋯<!-- ⋯ --></mo>
<mrow class="MJX-TeXAtom-ORD">
<mtext>&nbsp;and&nbsp;</mtext>
</mrow>
<msubsup>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
<mo>,</mo>
<mi>K</mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mo>∗<!-- ∗ --></mo>
</mrow>
</msubsup>
<mo>&gt;</mo>
<msubsup>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
<mo>,</mo>
<mi>K</mi>
<mo>−<!-- − --></mo>
<mn>1</mn>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mo>∗<!-- ∗ --></mo>
</mrow>
</msubsup>
<mo stretchy="false">)</mo>
</mtd>
</mtr>
</mtable>
</mrow>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle {\begin{aligned}\Pr(Y_{i}=1)&amp;=\Pr(Y_{i,1}^{\ast }&gt;Y_{i,2}^{\ast }{\text{ and }}Y_{i,1}^{\ast }&gt;Y_{i,3}^{\ast }{\text{ and }}\cdots {\text{ and }}Y_{i,1}^{\ast }&gt;Y_{i,K}^{\ast })\\\Pr(Y_{i}=2)&amp;=\Pr(Y_{i,2}^{\ast }&gt;Y_{i,1}^{\ast }{\text{ and }}Y_{i,2}^{\ast }&gt;Y_{i,3}^{\ast }{\text{ and }}\cdots {\text{ and }}Y_{i,2}^{\ast }&gt;Y_{i,K}^{\ast })\\&amp;\,\,\,\vdots \\\Pr(Y_{i}=K)&amp;=\Pr(Y_{i,K}^{\ast }&gt;Y_{i,1}^{\ast }{\text{ and }}Y_{i,K}^{\ast }&gt;Y_{i,2}^{\ast }{\text{ and }}\cdots {\text{ and }}Y_{i,K}^{\ast }&gt;Y_{i,K-1}^{\ast })\\\end{aligned}}}</annotation>
</semantics>
</math></span><img src="./8e280be3f36649906431730edfcf92e08823779b.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -6.838ex; width:72.525ex; height:14.843ex;" alt="{\displaystyle {\begin{aligned}\Pr(Y_{i}=1)&amp;=\Pr(Y_{i,1}^{\ast }>Y_{i,2}^{\ast }{\text{ and }}Y_{i,1}^{\ast }>Y_{i,3}^{\ast }{\text{ and }}\cdots {\text{ and }}Y_{i,1}^{\ast }>Y_{i,K}^{\ast })\\\Pr(Y_{i}=2)&amp;=\Pr(Y_{i,2}^{\ast }>Y_{i,1}^{\ast }{\text{ and }}Y_{i,2}^{\ast }>Y_{i,3}^{\ast }{\text{ and }}\cdots {\text{ and }}Y_{i,2}^{\ast }>Y_{i,K}^{\ast })\\&amp;\,\,\,\vdots \\\Pr(Y_{i}=K)&amp;=\Pr(Y_{i,K}^{\ast }>Y_{i,1}^{\ast }{\text{ and }}Y_{i,K}^{\ast }>Y_{i,2}^{\ast }{\text{ and }}\cdots {\text{ and }}Y_{i,K}^{\ast }>Y_{i,K-1}^{\ast })\\\end{aligned}}}" loading="lazy"></span></dd></dl>
<p>Or equivalently:
</p>
<dl><dd><span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle \Pr(Y_{i}=k)\;=\;\Pr(\max(Y_{i,1}^{\ast },Y_{i,2}^{\ast },\ldots ,Y_{i,K}^{\ast })=Y_{i,k}^{\ast })\;\;\;\;,\;\;k\leq K}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mo movablelimits="true" form="prefix">Pr</mo>
<mo stretchy="false">(</mo>
<msub>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
<mo>=</mo>
<mi>k</mi>
<mo stretchy="false">)</mo>
<mspace width="thickmathspace"></mspace>
<mo>=</mo>
<mspace width="thickmathspace"></mspace>
<mo movablelimits="true" form="prefix">Pr</mo>
<mo stretchy="false">(</mo>
<mo movablelimits="true" form="prefix">max</mo>
<mo stretchy="false">(</mo>
<msubsup>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
<mo>,</mo>
<mn>1</mn>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mo>∗<!-- ∗ --></mo>
</mrow>
</msubsup>
<mo>,</mo>
<msubsup>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
<mo>,</mo>
<mn>2</mn>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mo>∗<!-- ∗ --></mo>
</mrow>
</msubsup>
<mo>,</mo>
<mo>…<!-- … --></mo>
<mo>,</mo>
<msubsup>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
<mo>,</mo>
<mi>K</mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mo>∗<!-- ∗ --></mo>
</mrow>
</msubsup>
<mo stretchy="false">)</mo>
<mo>=</mo>
<msubsup>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
<mo>,</mo>
<mi>k</mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mo>∗<!-- ∗ --></mo>
</mrow>
</msubsup>
<mo stretchy="false">)</mo>
<mspace width="thickmathspace"></mspace>
<mspace width="thickmathspace"></mspace>
<mspace width="thickmathspace"></mspace>
<mspace width="thickmathspace"></mspace>
<mo>,</mo>
<mspace width="thickmathspace"></mspace>
<mspace width="thickmathspace"></mspace>
<mi>k</mi>
<mo>≤<!-- ≤ --></mo>
<mi>K</mi>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle \Pr(Y_{i}=k)\;=\;\Pr(\max(Y_{i,1}^{\ast },Y_{i,2}^{\ast },\ldots ,Y_{i,K}^{\ast })=Y_{i,k}^{\ast })\;\;\;\;,\;\;k\leq K}</annotation>
</semantics>
</math></span><img src="./85e9b395d4d39883caedef6ae792e26b2bc9ef53.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -1.338ex; width:60.574ex; height:3.343ex;" alt="{\displaystyle \Pr(Y_{i}=k)\;=\;\Pr(\max(Y_{i,1}^{\ast },Y_{i,2}^{\ast },\ldots ,Y_{i,K}^{\ast })=Y_{i,k}^{\ast })\;\;\;\;,\;\;k\leq K}" loading="lazy"></span></dd></dl>
<p>Let's look more closely at the first equation, which we can write as follows:
</p>
<dl><dd><span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle {\begin{aligned}\Pr(Y_{i}=1)&amp;=\Pr(Y_{i,1}^{\ast }>Y_{i,k}^{\ast }\ \forall \ k=2,\ldots ,K)\\&amp;=\Pr(Y_{i,1}^{\ast }-Y_{i,k}^{\ast }>0\ \forall \ k=2,\ldots ,K)\\&amp;=\Pr({\boldsymbol {\beta }}_{1}\cdot \mathbf {X} _{i}+\varepsilon _{1}-({\boldsymbol {\beta }}_{k}\cdot \mathbf {X} _{i}+\varepsilon _{k})>0\ \forall \ k=2,\ldots ,K)\\&amp;=\Pr(({\boldsymbol {\beta }}_{1}-{\boldsymbol {\beta }}_{k})\cdot \mathbf {X} _{i}>\varepsilon _{k}-\varepsilon _{1}\ \forall \ k=2,\ldots ,K)\end{aligned}}}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mrow class="MJX-TeXAtom-ORD">
<mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true">
<mtr>
<mtd>
<mo movablelimits="true" form="prefix">Pr</mo>
<mo stretchy="false">(</mo>
<msub>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
<mo>=</mo>
<mn>1</mn>
<mo stretchy="false">)</mo>
</mtd>
<mtd>
<mi></mi>
<mo>=</mo>
<mo movablelimits="true" form="prefix">Pr</mo>
<mo stretchy="false">(</mo>
<msubsup>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
<mo>,</mo>
<mn>1</mn>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mo>∗<!-- ∗ --></mo>
</mrow>
</msubsup>
<mo>&gt;</mo>
<msubsup>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
<mo>,</mo>
<mi>k</mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mo>∗<!-- ∗ --></mo>
</mrow>
</msubsup>
<mtext>&nbsp;</mtext>
<mi mathvariant="normal">∀<!-- ∀ --></mi>
<mtext>&nbsp;</mtext>
<mi>k</mi>
<mo>=</mo>
<mn>2</mn>
<mo>,</mo>
<mo>…<!-- … --></mo>
<mo>,</mo>
<mi>K</mi>
<mo stretchy="false">)</mo>
</mtd>
</mtr>
<mtr>
<mtd></mtd>
<mtd>
<mi></mi>
<mo>=</mo>
<mo movablelimits="true" form="prefix">Pr</mo>
<mo stretchy="false">(</mo>
<msubsup>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
<mo>,</mo>
<mn>1</mn>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mo>∗<!-- ∗ --></mo>
</mrow>
</msubsup>
<mo>−<!-- − --></mo>
<msubsup>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
<mo>,</mo>
<mi>k</mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mo>∗<!-- ∗ --></mo>
</mrow>
</msubsup>
<mo>&gt;</mo>
<mn>0</mn>
<mtext>&nbsp;</mtext>
<mi mathvariant="normal">∀<!-- ∀ --></mi>
<mtext>&nbsp;</mtext>
<mi>k</mi>
<mo>=</mo>
<mn>2</mn>
<mo>,</mo>
<mo>…<!-- … --></mo>
<mo>,</mo>
<mi>K</mi>
<mo stretchy="false">)</mo>
</mtd>
</mtr>
<mtr>
<mtd></mtd>
<mtd>
<mi></mi>
<mo>=</mo>
<mo movablelimits="true" form="prefix">Pr</mo>
<mo stretchy="false">(</mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold-italic">β<!-- β --></mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mn>1</mn>
</mrow>
</msub>
<mo>⋅<!-- ⋅ --></mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold">X</mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
<mo>+</mo>
<msub>
<mi>ε<!-- ε --></mi>
<mrow class="MJX-TeXAtom-ORD">
<mn>1</mn>
</mrow>
</msub>
<mo>−<!-- − --></mo>
<mo stretchy="false">(</mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold-italic">β<!-- β --></mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>k</mi>
</mrow>
</msub>
<mo>⋅<!-- ⋅ --></mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold">X</mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
<mo>+</mo>
<msub>
<mi>ε<!-- ε --></mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>k</mi>
</mrow>
</msub>
<mo stretchy="false">)</mo>
<mo>&gt;</mo>
<mn>0</mn>
<mtext>&nbsp;</mtext>
<mi mathvariant="normal">∀<!-- ∀ --></mi>
<mtext>&nbsp;</mtext>
<mi>k</mi>
<mo>=</mo>
<mn>2</mn>
<mo>,</mo>
<mo>…<!-- … --></mo>
<mo>,</mo>
<mi>K</mi>
<mo stretchy="false">)</mo>
</mtd>
</mtr>
<mtr>
<mtd></mtd>
<mtd>
<mi></mi>
<mo>=</mo>
<mo movablelimits="true" form="prefix">Pr</mo>
<mo stretchy="false">(</mo>
<mo stretchy="false">(</mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold-italic">β<!-- β --></mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mn>1</mn>
</mrow>
</msub>
<mo>−<!-- − --></mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold-italic">β<!-- β --></mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>k</mi>
</mrow>
</msub>
<mo stretchy="false">)</mo>
<mo>⋅<!-- ⋅ --></mo>
<msub>
<mrow class="MJX-TeXAtom-ORD">
<mi mathvariant="bold">X</mi>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
<mo>&gt;</mo>
<msub>
<mi>ε<!-- ε --></mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>k</mi>
</mrow>
</msub>
<mo>−<!-- − --></mo>
<msub>
<mi>ε<!-- ε --></mi>
<mrow class="MJX-TeXAtom-ORD">
<mn>1</mn>
</mrow>
</msub>
<mtext>&nbsp;</mtext>
<mi mathvariant="normal">∀<!-- ∀ --></mi>
<mtext>&nbsp;</mtext>
<mi>k</mi>
<mo>=</mo>
<mn>2</mn>
<mo>,</mo>
<mo>…<!-- … --></mo>
<mo>,</mo>
<mi>K</mi>
<mo stretchy="false">)</mo>
</mtd>
</mtr>
</mtable>
</mrow>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle {\begin{aligned}\Pr(Y_{i}=1)&amp;=\Pr(Y_{i,1}^{\ast }&gt;Y_{i,k}^{\ast }\ \forall \ k=2,\ldots ,K)\\&amp;=\Pr(Y_{i,1}^{\ast }-Y_{i,k}^{\ast }&gt;0\ \forall \ k=2,\ldots ,K)\\&amp;=\Pr({\boldsymbol {\beta }}_{1}\cdot \mathbf {X} _{i}+\varepsilon _{1}-({\boldsymbol {\beta }}_{k}\cdot \mathbf {X} _{i}+\varepsilon _{k})&gt;0\ \forall \ k=2,\ldots ,K)\\&amp;=\Pr(({\boldsymbol {\beta }}_{1}-{\boldsymbol {\beta }}_{k})\cdot \mathbf {X} _{i}&gt;\varepsilon _{k}-\varepsilon _{1}\ \forall \ k=2,\ldots ,K)\end{aligned}}}</annotation>
</semantics>
</math></span><img src="./e960e5129f14aeb19596491b4fa8ece11c3e6149.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -6.171ex; width:67.148ex; height:13.509ex;" alt="{\displaystyle {\begin{aligned}\Pr(Y_{i}=1)&amp;=\Pr(Y_{i,1}^{\ast }>Y_{i,k}^{\ast }\ \forall \ k=2,\ldots ,K)\\&amp;=\Pr(Y_{i,1}^{\ast }-Y_{i,k}^{\ast }>0\ \forall \ k=2,\ldots ,K)\\&amp;=\Pr({\boldsymbol {\beta }}_{1}\cdot \mathbf {X} _{i}+\varepsilon _{1}-({\boldsymbol {\beta }}_{k}\cdot \mathbf {X} _{i}+\varepsilon _{k})>0\ \forall \ k=2,\ldots ,K)\\&amp;=\Pr(({\boldsymbol {\beta }}_{1}-{\boldsymbol {\beta }}_{k})\cdot \mathbf {X} _{i}>\varepsilon _{k}-\varepsilon _{1}\ \forall \ k=2,\ldots ,K)\end{aligned}}}" loading="lazy"></span></dd></dl>
<p>There are a few things to realize here:
</p>
<ol><li>In general, if <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle X\sim \operatorname {EV} _{1}(a,b)}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>X</mi>
<mo>∼<!-- ∼ --></mo>
<msub>
<mi>EV</mi>
<mrow class="MJX-TeXAtom-ORD">
<mn>1</mn>
</mrow>
</msub>
<mo>⁡<!-- ⁡ --></mo>
<mo stretchy="false">(</mo>
<mi>a</mi>
<mo>,</mo>
<mi>b</mi>
<mo stretchy="false">)</mo>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle X\sim \operatorname {EV} _{1}(a,b)}</annotation>
</semantics>
</math></span><img src="./b950c0d339f29af3db351aca73db6d24f9cb6b68.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.838ex; width:14.529ex; height:2.843ex;" alt="{\displaystyle X\sim \operatorname {EV} _{1}(a,b)}" loading="lazy"></span> and <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle Y\sim \operatorname {EV} _{1}(a,b)}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>Y</mi>
<mo>∼<!-- ∼ --></mo>
<msub>
<mi>EV</mi>
<mrow class="MJX-TeXAtom-ORD">
<mn>1</mn>
</mrow>
</msub>
<mo>⁡<!-- ⁡ --></mo>
<mo stretchy="false">(</mo>
<mi>a</mi>
<mo>,</mo>
<mi>b</mi>
<mo stretchy="false">)</mo>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle Y\sim \operatorname {EV} _{1}(a,b)}</annotation>
</semantics>
</math></span><img src="./c1e23aae9481579541829efafdedca1839b6bb88.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.838ex; width:14.323ex; height:2.843ex;" alt="{\displaystyle Y\sim \operatorname {EV} _{1}(a,b)}" loading="lazy"></span> then <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle X-Y\sim \operatorname {Logistic} (0,b).}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>X</mi>
<mo>−<!-- − --></mo>
<mi>Y</mi>
<mo>∼<!-- ∼ --></mo>
<mi>Logistic</mi>
<mo>⁡<!-- ⁡ --></mo>
<mo stretchy="false">(</mo>
<mn>0</mn>
<mo>,</mo>
<mi>b</mi>
<mo stretchy="false">)</mo>
<mo>.</mo>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle X-Y\sim \operatorname {Logistic} (0,b).}</annotation>
</semantics>
</math></span><img src="./8738abee03d0702495cce4b29b3ff76d9468e03a.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.838ex; width:23.267ex; height:2.843ex;" alt="{\displaystyle X-Y\sim \operatorname {Logistic} (0,b).}" loading="lazy"></span> That is, the difference of two <a href="Independent_identically_distributed" class="mw-redirect" title="Independent identically distributed">independent identically distributed</a> extreme-value-distributed variables follows the <a href="Logistic_distribution" title="Logistic distribution">logistic distribution</a>, where the first parameter is unimportant. This is understandable since the first parameter is a <a href="Location_parameter" title="Location parameter">location parameter</a>, i.e. it shifts the mean by a fixed amount, and if two values are both shifted by the same amount, their difference remains the same. This means that all of the relational statements underlying the probability of a given choice involve the logistic distribution, which makes the initial choice of the extreme-value distribution, which seemed rather arbitrary, somewhat more understandable.</li>
<li>The second parameter in an extreme-value or logistic distribution is a <a href="Scale_parameter" title="Scale parameter">scale parameter</a>, such that if <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle X\sim \operatorname {Logistic} (0,1)}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>X</mi>
<mo>∼<!-- ∼ --></mo>
<mi>Logistic</mi>
<mo>⁡<!-- ⁡ --></mo>
<mo stretchy="false">(</mo>
<mn>0</mn>
<mo>,</mo>
<mn>1</mn>
<mo stretchy="false">)</mo>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle X\sim \operatorname {Logistic} (0,1)}</annotation>
</semantics>
</math></span><img src="./30b3f76d845425eba3374505a198c1d1a1ccc96c.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.838ex; width:18.171ex; height:2.843ex;" alt="{\displaystyle X\sim \operatorname {Logistic} (0,1)}" loading="lazy"></span> then <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle bX\sim \operatorname {Logistic} (0,b).}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>b</mi>
<mi>X</mi>
<mo>∼<!-- ∼ --></mo>
<mi>Logistic</mi>
<mo>⁡<!-- ⁡ --></mo>
<mo stretchy="false">(</mo>
<mn>0</mn>
<mo>,</mo>
<mi>b</mi>
<mo stretchy="false">)</mo>
<mo>.</mo>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle bX\sim \operatorname {Logistic} (0,b).}</annotation>
</semantics>
</math></span><img src="./4a45ea113ea95dfc3b5a2cd907c05a266e766b50.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.838ex; width:19.651ex; height:2.843ex;" alt="{\displaystyle bX\sim \operatorname {Logistic} (0,b).}" loading="lazy"></span> This means that the effect of using an error variable with an arbitrary scale parameter in place of scale 1 can be compensated simply by multiplying all regression vectors by the same scale. Together with the previous point, this shows that the use of a standard extreme-value distribution (location 0, scale 1) for the error variables entails no loss of generality over using an arbitrary extreme-value distribution. In fact, the model is <a href="Nonidentifiable" class="mw-redirect" title="Nonidentifiable">nonidentifiable</a> (no single set of optimal coefficients) if the more general distribution is used.</li>
<li>Because only differences of vectors of regression coefficients are used, adding an arbitrary constant to all coefficient vectors has no effect on the model. This means that, just as in the log-linear model, only <i>K</i>&nbsp;−&nbsp;1 of the coefficient vectors are identifiable, and the last one can be set to an arbitrary value (e.g. 0).</li></ol>
<p>Actually finding the values of the above probabilities is somewhat difficult, and is a problem of computing a particular <a href="Order_statistic" title="Order statistic">order statistic</a> (the first, i.e. maximum) of a set of values. However, it can be shown that the resulting expressions are the same as in above formulations, i.e. the two are equivalent.
</p>
<div class="mw-heading mw-heading2"><h2 id="Estimation_of_intercept">Estimation of intercept</h2></div>
<p>When using multinomial logistic regression, one category of the dependent variable is chosen as the reference category. Separate <a href="Odds_ratio" title="Odds ratio">odds ratios</a> are determined for all independent variables for each category of the dependent variable with the exception of the reference category, which is omitted from the analysis. The exponential beta coefficient represents the change in the odds of the dependent variable being in a particular category vis-a-vis the reference category, associated with a one unit change of the corresponding independent variable.
</p>
<div class="mw-heading mw-heading2"><h2 id="Likelihood_function">Likelihood function</h2></div>
<p>The observed values <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle y_{i}\in \{1,\dots ,K\}}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<msub>
<mi>y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
<mo>∈<!-- ∈ --></mo>
<mo fence="false" stretchy="false">{</mo>
<mn>1</mn>
<mo>,</mo>
<mo>…<!-- … --></mo>
<mo>,</mo>
<mi>K</mi>
<mo fence="false" stretchy="false">}</mo>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle y_{i}\in \{1,\dots ,K\}}</annotation>
</semantics>
</math></span><img src="./ff0defa670b42083fb95a36b895a2a2e2613c04c.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.838ex; width:15.511ex; height:2.843ex;" alt="{\displaystyle y_{i}\in \{1,\dots ,K\}}" loading="lazy"></span> for <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle i=1,\dots ,n}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>i</mi>
<mo>=</mo>
<mn>1</mn>
<mo>,</mo>
<mo>…<!-- … --></mo>
<mo>,</mo>
<mi>n</mi>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle i=1,\dots ,n}</annotation>
</semantics>
</math></span><img src="./f3f269b2f3b2f87fec0168426652a5ea80b56112.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.671ex; width:11.636ex; height:2.509ex;" alt="{\displaystyle i=1,\dots ,n}" loading="lazy"></span> of the explained variables are considered as realizations of stochastically independent, <a href="Categorical_distribution" title="Categorical distribution">categorically distributed</a> random variables <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle Y_{1},\dots ,Y_{n}}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<msub>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mn>1</mn>
</mrow>
</msub>
<mo>,</mo>
<mo>…<!-- … --></mo>
<mo>,</mo>
<msub>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>n</mi>
</mrow>
</msub>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle Y_{1},\dots ,Y_{n}}</annotation>
</semantics>
</math></span><img src="./010c105dda336a4624a635ea54886fb040034d64.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.671ex; width:10.152ex; height:2.509ex;" alt="{\displaystyle Y_{1},\dots ,Y_{n}}" loading="lazy"></span>.
</p><p>The <a href="Likelihood_function" title="Likelihood function">likelihood function</a> for this model is defined by
</p>
<dl><dd><span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle L=\prod _{i=1}^{n}P(Y_{i}=y_{i})=\prod _{i=1}^{n}\prod _{j=1}^{K}P(Y_{i}=j)^{\delta _{j,y_{i}}},}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>L</mi>
<mo>=</mo>
<munderover>
<mo>∏<!-- ∏ --></mo>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
<mo>=</mo>
<mn>1</mn>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>n</mi>
</mrow>
</munderover>
<mi>P</mi>
<mo stretchy="false">(</mo>
<msub>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
<mo>=</mo>
<msub>
<mi>y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
<mo stretchy="false">)</mo>
<mo>=</mo>
<munderover>
<mo>∏<!-- ∏ --></mo>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
<mo>=</mo>
<mn>1</mn>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>n</mi>
</mrow>
</munderover>
<munderover>
<mo>∏<!-- ∏ --></mo>
<mrow class="MJX-TeXAtom-ORD">
<mi>j</mi>
<mo>=</mo>
<mn>1</mn>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>K</mi>
</mrow>
</munderover>
<mi>P</mi>
<mo stretchy="false">(</mo>
<msub>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
<mo>=</mo>
<mi>j</mi>
<msup>
<mo stretchy="false">)</mo>
<mrow class="MJX-TeXAtom-ORD">
<msub>
<mi>δ<!-- δ --></mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>j</mi>
<mo>,</mo>
<msub>
<mi>y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
</mrow>
</msub>
</mrow>
</msup>
<mo>,</mo>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle L=\prod _{i=1}^{n}P(Y_{i}=y_{i})=\prod _{i=1}^{n}\prod _{j=1}^{K}P(Y_{i}=j)^{\delta _{j,y_{i}}},}</annotation>
</semantics>
</math></span><img src="./4aeb064328b9d49bb2a6ba35486bed10fb4a178b.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -3.338ex; width:42.296ex; height:7.676ex;" alt="{\displaystyle L=\prod _{i=1}^{n}P(Y_{i}=y_{i})=\prod _{i=1}^{n}\prod _{j=1}^{K}P(Y_{i}=j)^{\delta _{j,y_{i}}},}" loading="lazy"></span></dd></dl>
<p>where the index <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle i}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>i</mi>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle i}</annotation>
</semantics>
</math></span><img src="./add78d8608ad86e54951b8c8bd6c8d8416533d20.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.338ex; width:0.802ex; height:2.176ex;" alt="{\displaystyle i}" loading="lazy"></span> denotes the observations 1 to <i>n</i> and the index <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle j}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>j</mi>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle j}</annotation>
</semantics>
</math></span><img src="./2f461e54f5c093e92a55547b9764291390f0b5d0.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.671ex; margin-left: -0.027ex; width:0.985ex; height:2.509ex;" alt="{\displaystyle j}" loading="lazy"></span> denotes the classes 1 to <i>K</i>. <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle \delta _{j,y_{i}}={\begin{cases}1,{\text{ for }}j=y_{i}\\0,{\text{ otherwise}}\end{cases}}}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<msub>
<mi>δ<!-- δ --></mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>j</mi>
<mo>,</mo>
<msub>
<mi>y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
</mrow>
</msub>
<mo>=</mo>
<mrow class="MJX-TeXAtom-ORD">
<mrow>
<mo>{</mo>
<mtable columnalign="left left" rowspacing=".2em" columnspacing="1em" displaystyle="false">
<mtr>
<mtd>
<mn>1</mn>
<mo>,</mo>
<mrow class="MJX-TeXAtom-ORD">
<mtext>&nbsp;for&nbsp;</mtext>
</mrow>
<mi>j</mi>
<mo>=</mo>
<msub>
<mi>y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
</mtd>
</mtr>
<mtr>
<mtd>
<mn>0</mn>
<mo>,</mo>
<mrow class="MJX-TeXAtom-ORD">
<mtext>&nbsp;otherwise</mtext>
</mrow>
</mtd>
</mtr>
</mtable>
<mo fence="true" stretchy="true" symmetric="true"></mo>
</mrow>
</mrow>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle \delta _{j,y_{i}}={\begin{cases}1,{\text{ for }}j=y_{i}\\0,{\text{ otherwise}}\end{cases}}}</annotation>
</semantics>
</math></span><img src="./43c1083e09653bacea524a526c7c37dfe963e6fa.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -2.505ex; width:21.777ex; height:6.176ex;" alt="{\displaystyle \delta _{j,y_{i}}={\begin{cases}1,{\text{ for }}j=y_{i}\\0,{\text{ otherwise}}\end{cases}}}" loading="lazy"></span> is the <a href="Kronecker_delta" title="Kronecker delta">Kronecker delta</a>.
</p><p>The negative log-likelihood function is therefore the well-known cross-entropy:
</p>
<dl><dd><span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle -\log L=-\sum _{i=1}^{n}\sum _{j=1}^{K}\delta _{j,y_{i}}\log(P(Y_{i}=j))=-\sum _{j=1}^{K}\sum _{y_{i}=j}\log(P(Y_{i}=j)).}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mo>−<!-- − --></mo>
<mi>log</mi>
<mo>⁡<!-- ⁡ --></mo>
<mi>L</mi>
<mo>=</mo>
<mo>−<!-- − --></mo>
<munderover>
<mo>∑<!-- ∑ --></mo>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
<mo>=</mo>
<mn>1</mn>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>n</mi>
</mrow>
</munderover>
<munderover>
<mo>∑<!-- ∑ --></mo>
<mrow class="MJX-TeXAtom-ORD">
<mi>j</mi>
<mo>=</mo>
<mn>1</mn>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>K</mi>
</mrow>
</munderover>
<msub>
<mi>δ<!-- δ --></mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>j</mi>
<mo>,</mo>
<msub>
<mi>y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
</mrow>
</msub>
<mi>log</mi>
<mo>⁡<!-- ⁡ --></mo>
<mo stretchy="false">(</mo>
<mi>P</mi>
<mo stretchy="false">(</mo>
<msub>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
<mo>=</mo>
<mi>j</mi>
<mo stretchy="false">)</mo>
<mo stretchy="false">)</mo>
<mo>=</mo>
<mo>−<!-- − --></mo>
<munderover>
<mo>∑<!-- ∑ --></mo>
<mrow class="MJX-TeXAtom-ORD">
<mi>j</mi>
<mo>=</mo>
<mn>1</mn>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<mi>K</mi>
</mrow>
</munderover>
<munder>
<mo>∑<!-- ∑ --></mo>
<mrow class="MJX-TeXAtom-ORD">
<msub>
<mi>y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
<mo>=</mo>
<mi>j</mi>
</mrow>
</munder>
<mi>log</mi>
<mo>⁡<!-- ⁡ --></mo>
<mo stretchy="false">(</mo>
<mi>P</mi>
<mo stretchy="false">(</mo>
<msub>
<mi>Y</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
<mo>=</mo>
<mi>j</mi>
<mo stretchy="false">)</mo>
<mo stretchy="false">)</mo>
<mo>.</mo>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle -\log L=-\sum _{i=1}^{n}\sum _{j=1}^{K}\delta _{j,y_{i}}\log(P(Y_{i}=j))=-\sum _{j=1}^{K}\sum _{y_{i}=j}\log(P(Y_{i}=j)).}</annotation>
</semantics>
</math></span><img src="./ecdc0f26f2870df9c165eaa100fd0425f6511c8c.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -3.505ex; width:66.673ex; height:7.843ex;" alt="{\displaystyle -\log L=-\sum _{i=1}^{n}\sum _{j=1}^{K}\delta _{j,y_{i}}\log(P(Y_{i}=j))=-\sum _{j=1}^{K}\sum _{y_{i}=j}\log(P(Y_{i}=j)).}" loading="lazy"></span></dd></dl>
<div class="mw-heading mw-heading2"><h2 id="Application_in_natural_language_processing">Application in natural language processing</h2></div>
<p>In <a href="Natural_language_processing" title="Natural language processing">natural language processing</a>, multinomial LR classifiers are commonly used as an alternative to <a href="Naive_Bayes_classifier" title="Naive Bayes classifier">naive Bayes classifiers</a> because they do not assume <a href="Statistical_independence" class="mw-redirect" title="Statistical independence">statistical independence</a> of the random variables (commonly known as <i>features</i>) that serve as predictors. However, learning in such a model is slower than for a naive Bayes classifier, and thus may not be appropriate given a very large number of classes to learn. In particular, learning in a naive Bayes classifier is a simple matter of counting up the number of co-occurrences of features and classes, while in a maximum entropy classifier the weights, which are typically maximized using <a href="Maximum_a_posteriori" class="mw-redirect" title="Maximum a posteriori">maximum a posteriori</a> (MAP) estimation, must be learned using an iterative procedure; see <a href="#Estimating_the_coefficients">#Estimating the coefficients</a>.
</p>
<div class="mw-heading mw-heading2"><h2 id="See_also">See also</h2></div>
<ul><li><a href="Logistic_regression" title="Logistic regression">Logistic regression</a></li>
<li><a href="Multinomial_probit" title="Multinomial probit">Multinomial probit</a></li></ul>
<div class="mw-heading mw-heading2"><h2 id="References">References</h2></div>
<style data-mw-deduplicate="TemplateStyles:r1239543626">
/* start https://en.wikipedia.org/ */


.mw-parser-output .reflist{margin-bottom:0.5em;list-style-type:decimal}@media screen{.mw-parser-output .reflist{font-size:90%}}.mw-parser-output .reflist .references{font-size:100%;margin-bottom:0;list-style-type:inherit}.mw-parser-output .reflist-columns-2{column-width:30em}.mw-parser-output .reflist-columns-3{column-width:25em}.mw-parser-output .reflist-columns{margin-top:0.3em}.mw-parser-output .reflist-columns ol{margin-top:0}.mw-parser-output .reflist-columns li{page-break-inside:avoid;break-inside:avoid-column}.mw-parser-output .reflist-upper-alpha{list-style-type:upper-alpha}.mw-parser-output .reflist-upper-roman{list-style-type:upper-roman}.mw-parser-output .reflist-lower-alpha{list-style-type:lower-alpha}.mw-parser-output .reflist-lower-greek{list-style-type:lower-greek}.mw-parser-output .reflist-lower-roman{list-style-type:lower-roman}


/* end https://en.wikipedia.org/ */
</style><div class="reflist reflist-columns references-column-width" style="column-width: 30em;">
<ol class="references">
<li id="cite_note-1"><span class="mw-cite-backlink"><b><a href="#cite_ref-1">^</a></b></span> <span class="reference-text"><style data-mw-deduplicate="TemplateStyles:r1238218222">
/* start https://en.wikipedia.org/ */


.mw-parser-output cite.citation{font-style:inherit;word-wrap:break-word}.mw-parser-output .citation q{quotes:"\"""\"""'""'"}.mw-parser-output .citation:target{background-color:rgba(0,127,255,0.133)}.mw-parser-output .id-lock-free.id-lock-free a{background:url("./mw/Lock-green.svg")right 0.1em center/9px no-repeat}.mw-parser-output .id-lock-limited.id-lock-limited a,.mw-parser-output .id-lock-registration.id-lock-registration a{background:url("./mw/Lock-gray-alt-2.svg")right 0.1em center/9px no-repeat}.mw-parser-output .id-lock-subscription.id-lock-subscription a{background:url("./mw/Lock-red-alt-2.svg")right 0.1em center/9px no-repeat}.mw-parser-output .cs1-ws-icon a{background:url("./mw/Wikisource-logo.svg")right 0.1em center/12px no-repeat}body:not(.skin-timeless):not(.skin-minerva) .mw-parser-output .id-lock-free a,body:not(.skin-timeless):not(.skin-minerva) .mw-parser-output .id-lock-limited a,body:not(.skin-timeless):not(.skin-minerva) .mw-parser-output .id-lock-registration a,body:not(.skin-timeless):not(.skin-minerva) .mw-parser-output .id-lock-subscription a,body:not(.skin-timeless):not(.skin-minerva) .mw-parser-output .cs1-ws-icon a{background-size:contain;padding:0 1em 0 0}.mw-parser-output .cs1-code{color:inherit;background:inherit;border:none;padding:inherit}.mw-parser-output .cs1-hidden-error{display:none;color:var(--color-error,#d33)}.mw-parser-output .cs1-visible-error{color:var(--color-error,#d33)}.mw-parser-output .cs1-maint{display:none;color:#085;margin-left:0.3em}.mw-parser-output .cs1-kern-left{padding-left:0.2em}.mw-parser-output .cs1-kern-right{padding-right:0.2em}.mw-parser-output .citation .mw-selflink{font-weight:inherit}@media screen{.mw-parser-output .cs1-format{font-size:95%}html.skin-theme-clientpref-night .mw-parser-output .cs1-maint{color:#18911f}}@media screen and (prefers-color-scheme:dark){html.skin-theme-clientpref-os .mw-parser-output .cs1-maint{color:#18911f}}


/* end https://en.wikipedia.org/ */
</style><cite id="CITEREFGreene2012" class="citation book cs1"><a href="William_Greene_(economist)" title="William Greene (economist)">Greene, William H.</a> (2012). <i>Econometric Analysis</i> (Seventh&nbsp;ed.). Boston: Pearson Education. pp.&nbsp;<span class="nowrap">803–</span>806. <a href="ISBN_(identifier)" class="mw-redirect" title="ISBN (identifier)">ISBN</a>&nbsp;<bdi>978-0-273-75356-8</bdi>.</cite></span>
</li>
<li id="cite_note-2"><span class="mw-cite-backlink"><b><a href="#cite_ref-2">^</a></b></span> <span class="reference-text"><cite id="CITEREFEngel1988" class="citation journal cs1">Engel, J. (1988). "Polytomous logistic regression". <i>Statistica Neerlandica</i>. <b>42</b> (4): <span class="nowrap">233–</span>252. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<a rel="nofollow" class="external text" href="https://doi.org/10.1111%2Fj.1467-9574.1988.tb01238.x">10.1111/j.1467-9574.1988.tb01238.x</a>.</cite></span>
</li>
<li id="cite_note-3"><span class="mw-cite-backlink"><b><a href="#cite_ref-3">^</a></b></span> <span class="reference-text"><cite id="CITEREFMenard2002" class="citation book cs1">Menard, Scott (2002). <span class="id-lock-limited" title="Free access subject to limited trial, subscription normally required"><a rel="nofollow" class="external text" href="https://archive.org/details/appliedlogisticr00mena"><i>Applied Logistic Regression Analysis</i></a></span>. SAGE. p.&nbsp;<a rel="nofollow" class="external text" href="https://archive.org/details/appliedlogisticr00mena/page/n99">91</a>. <a href="ISBN_(identifier)" class="mw-redirect" title="ISBN (identifier)">ISBN</a>&nbsp;<bdi>9780761922087</bdi>.</cite></span>
</li>
<li id="cite_note-malouf-4"><span class="mw-cite-backlink">^ <a href="#cite_ref-malouf_4-0"><sup><i><b>a</b></i></sup></a> <a href="#cite_ref-malouf_4-1"><sup><i><b>b</b></i></sup></a></span> <span class="reference-text"><cite id="CITEREFMalouf2002" class="citation conference cs1">Malouf, Robert (2002). <a rel="nofollow" class="external text" href="http://aclweb.org/anthology/W/W02/W02-2018.pdf"><i>A comparison of algorithms for maximum entropy parameter estimation</i></a> <span class="cs1-format">(PDF)</span>. Sixth Conf. on Natural Language Learning (CoNLL). pp.&nbsp;<span class="nowrap">49–</span>55.</cite></span>
</li>
<li id="cite_note-5"><span class="mw-cite-backlink"><b><a href="#cite_ref-5">^</a></b></span> <span class="reference-text"><cite id="CITEREFBelsley1991" class="citation book cs1">Belsley, David (1991). <i>Conditioning diagnostics&nbsp;: collinearity and weak data in regression</i>. New York: Wiley. <a href="ISBN_(identifier)" class="mw-redirect" title="ISBN (identifier)">ISBN</a>&nbsp;<bdi>9780471528890</bdi>.</cite></span>
</li>
<li id="cite_note-6"><span class="mw-cite-backlink"><b><a href="#cite_ref-6">^</a></b></span> <span class="reference-text"><cite id="CITEREFBaltasDoyle2001" class="citation journal cs1">Baltas, G.; Doyle, P. (2001). "Random Utility Models in Marketing Research: A Survey". <i><a href="Journal_of_Business_Research" title="Journal of Business Research">Journal of Business Research</a></i>. <b>51</b> (2): <span class="nowrap">115–</span>125. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<a rel="nofollow" class="external text" href="https://doi.org/10.1016%2FS0148-2963%2899%2900058-2">10.1016/S0148-2963(99)00058-2</a>.</cite></span>
</li>
<li id="cite_note-7"><span class="mw-cite-backlink"><b><a href="#cite_ref-7">^</a></b></span> <span class="reference-text"><a rel="nofollow" class="external text" href="https://www.stata.com/manuals13/rmlogit.pdf">Stata Manual “mlogit — Multinomial (polytomous) logistic regression”</a></span>
</li>
<li id="cite_note-8"><span class="mw-cite-backlink"><b><a href="#cite_ref-8">^</a></b></span> <span class="reference-text"><cite id="CITEREFDarroch,_J.N.Ratcliff,_D.1972" class="citation journal cs1">Darroch, J.N. &amp; Ratcliff, D. (1972). <a rel="nofollow" class="external text" href="http://projecteuclid.org/download/pdf_1/euclid.aoms/1177692379">"Generalized iterative scaling for log-linear models"</a>. <i>The Annals of Mathematical Statistics</i>. <b>43</b> (5): <span class="nowrap">1470–</span>1480. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://doi.org/10.1214%2Faoms%2F1177692379">10.1214/aoms/1177692379</a></span>.</cite></span>
</li>
<li id="cite_note-9"><span class="mw-cite-backlink"><b><a href="#cite_ref-9">^</a></b></span> <span class="reference-text"><cite id="CITEREFBishop2006" class="citation book cs1">Bishop, Christopher M. (2006). <i>Pattern Recognition and Machine Learning</i>. Springer. pp.&nbsp;<span class="nowrap">206–</span>209.</cite></span>
</li>
<li id="cite_note-10"><span class="mw-cite-backlink"><b><a href="#cite_ref-10">^</a></b></span> <span class="reference-text"><cite id="CITEREFYuHuangLin2011" class="citation journal cs1">Yu, Hsiang-Fu; Huang, Fang-Lan; Lin, Chih-Jen (2011). <a rel="nofollow" class="external text" href="http://www.csie.ntu.edu.tw/~cjlin/papers/maxent_dual.pdf">"Dual coordinate descent methods for logistic regression and maximum entropy models"</a> <span class="cs1-format">(PDF)</span>. <i>Machine Learning</i>. <b>85</b> (<span class="nowrap">1–</span>2): <span class="nowrap">41–</span>75. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://doi.org/10.1007%2Fs10994-010-5221-8">10.1007/s10994-010-5221-8</a></span>.</cite></span>
</li>
</ol></div></div><!--htdig_noindex--><div><div class="zim-footer">
This article is issued from <a class="external text" title="Last edited on 2025-03-03" href="https://en.wikipedia.org/wiki/?title=Multinomial_logistic_regression&amp;oldid=1278597985">Wikipedia</a>. The text is available under <a class="external text" href="https://creativecommons.org/licenses/by-sa/4.0/deed.en">Creative Commons Attribution-Share Alike 4.0</a> unless otherwise noted. Additional terms may apply for the media files.
</div>
</div><!--/htdig_noindex--></div>
</div>
</main>
</div>
</div>
</div>

</body></html>